icu-rbnf 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,96 @@
1
+ Metadata-Version: 2.4
2
+ Name: icu_rbnf
3
+ Version: 0.2.0
4
+ Summary: Spell out numbers into words using ICU RBNF
5
+ Home-page: https://github.com/OHF-voice/icu-rbnf
6
+ Author: The Home Assistant Authors
7
+ Author-email: hello@home-assistant.io
8
+ Project-URL: Homepage, https://github.com/OHF-voice/icu-rbnf
9
+ Project-URL: Issues, https://github.com/OHF-voice/icu-rbnf/issues
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: License :: OSI Approved :: Apache Software License
13
+ Classifier: Operating System :: OS Independent
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.9
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Classifier: Programming Language :: C
21
+ Requires-Python: >=3.9
22
+ Description-Content-Type: text/markdown
23
+ Provides-Extra: dev
24
+ Requires-Dist: black; extra == "dev"
25
+ Requires-Dist: flake8; extra == "dev"
26
+ Requires-Dist: mypy; extra == "dev"
27
+ Requires-Dist: pylint; extra == "dev"
28
+ Requires-Dist: pytest; extra == "dev"
29
+ Requires-Dist: build; extra == "dev"
30
+ Dynamic: author
31
+ Dynamic: author-email
32
+ Dynamic: classifier
33
+ Dynamic: description
34
+ Dynamic: description-content-type
35
+ Dynamic: home-page
36
+ Dynamic: project-url
37
+ Dynamic: provides-extra
38
+ Dynamic: requires-python
39
+ Dynamic: summary
40
+
41
+ # icu_rbnf
42
+
43
+ A Python library for spelling out numbers into words using ICU's Rule-Based Number Format (RBNF).
44
+
45
+ ## Installation
46
+
47
+ ```bash
48
+ pip install icu_rbnf
49
+ ```
50
+
51
+ ## Usage
52
+
53
+ ```python
54
+ import icu_rbnf
55
+
56
+ # Spell out numbers in words
57
+ icu_rbnf.spellout(123, "en_US") # "one hundred twenty-three"
58
+ icu_rbnf.spellout(123.7, "en_US") # "one hundred twenty-three point seven"
59
+ icu_rbnf.spellout(123, "fr_FR") # "cent vingt-trois"
60
+
61
+ # Get ordinal form (e.g., "1st", "2nd")
62
+ icu_rbnf.ordinal(21, "en_US") # "21st"
63
+
64
+ # Get word-based ordinal (e.g., "first", "twenty-first")
65
+ icu_rbnf.spellout_ordinal(21, "en_US") # "twenty-first"
66
+
67
+ # Check if a locale is supported
68
+ icu_rbnf.is_locale_supported("en_US") # True
69
+ ```
70
+
71
+ ## API
72
+
73
+ ### `spellout(number: int | float, locale: str) -> str`
74
+
75
+ Spell out a number into words for the given locale. Supports both integers and floats.
76
+
77
+ ### `ordinal(number: int | float, locale: str) -> str`
78
+
79
+ Get the ordinal form of a number for the given locale (e.g., "1st", "2nd"). Floats are truncated to integers.
80
+
81
+ ### `spellout_ordinal(number: int | float, locale: str) -> str`
82
+
83
+ Spell out ordinal form of a number for the given locale (e.g., "first", "twenty-first"). Floats are truncated to integers. Note: Not all locales support word-based ordinals; some may fall back to numeric format.
84
+
85
+ ### `is_locale_supported(locale: str) -> bool`
86
+
87
+ Check if a locale is supported by ICU RBNF.
88
+
89
+ ### `error`
90
+
91
+ Exception class raised when ICU RBNF operations fail.
92
+
93
+ ## Requirements
94
+
95
+ - Python 3.9+
96
+ - ICU library (dynamically linked, no separate installation required)
@@ -0,0 +1,56 @@
1
+ # icu_rbnf
2
+
3
+ A Python library for spelling out numbers into words using ICU's Rule-Based Number Format (RBNF).
4
+
5
+ ## Installation
6
+
7
+ ```bash
8
+ pip install icu_rbnf
9
+ ```
10
+
11
+ ## Usage
12
+
13
+ ```python
14
+ import icu_rbnf
15
+
16
+ # Spell out numbers in words
17
+ icu_rbnf.spellout(123, "en_US") # "one hundred twenty-three"
18
+ icu_rbnf.spellout(123.7, "en_US") # "one hundred twenty-three point seven"
19
+ icu_rbnf.spellout(123, "fr_FR") # "cent vingt-trois"
20
+
21
+ # Get ordinal form (e.g., "1st", "2nd")
22
+ icu_rbnf.ordinal(21, "en_US") # "21st"
23
+
24
+ # Get word-based ordinal (e.g., "first", "twenty-first")
25
+ icu_rbnf.spellout_ordinal(21, "en_US") # "twenty-first"
26
+
27
+ # Check if a locale is supported
28
+ icu_rbnf.is_locale_supported("en_US") # True
29
+ ```
30
+
31
+ ## API
32
+
33
+ ### `spellout(number: int | float, locale: str) -> str`
34
+
35
+ Spell out a number into words for the given locale. Supports both integers and floats.
36
+
37
+ ### `ordinal(number: int | float, locale: str) -> str`
38
+
39
+ Get the ordinal form of a number for the given locale (e.g., "1st", "2nd"). Floats are truncated to integers.
40
+
41
+ ### `spellout_ordinal(number: int | float, locale: str) -> str`
42
+
43
+ Spell out ordinal form of a number for the given locale (e.g., "first", "twenty-first"). Floats are truncated to integers. Note: Not all locales support word-based ordinals; some may fall back to numeric format.
44
+
45
+ ### `is_locale_supported(locale: str) -> bool`
46
+
47
+ Check if a locale is supported by ICU RBNF.
48
+
49
+ ### `error`
50
+
51
+ Exception class raised when ICU RBNF operations fail.
52
+
53
+ ## Requirements
54
+
55
+ - Python 3.9+
56
+ - ICU library (dynamically linked, no separate installation required)
@@ -0,0 +1,22 @@
1
+ """ICU RBNF - Spell out numbers into words using ICU's Rule-Based Number Format."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from icu_rbnf._icu import (
6
+ error,
7
+ icu_version,
8
+ is_locale_supported,
9
+ ordinal,
10
+ spellout,
11
+ spellout_ordinal,
12
+ )
13
+
14
+ __all__ = [
15
+ "spellout",
16
+ "ordinal",
17
+ "spellout_ordinal",
18
+ "is_locale_supported",
19
+ "icu_version",
20
+ "error",
21
+ ]
22
+ __version__ = "0.2.0"
@@ -0,0 +1,28 @@
1
+ """Type stubs for icu_rbnf."""
2
+
3
+ from typing import Union
4
+
5
+ class RBNFError(Exception):
6
+ """Exception raised when ICU RBNF operations fail."""
7
+
8
+ def spellout(number: Union[int, float], locale: str) -> str:
9
+ """Spell out a number into words for the given locale."""
10
+ ...
11
+
12
+ def ordinal(number: Union[int, float], locale: str) -> str:
13
+ """Get the ordinal form of a number for the given locale (e.g., '1st', '2nd')."""
14
+ ...
15
+
16
+ def spellout_ordinal(number: Union[int, float], locale: str) -> str:
17
+ """Spell out ordinal form of a number for the given locale (e.g., 'first', 'twenty-first')."""
18
+ ...
19
+
20
+ def icu_version() -> str:
21
+ """Version of the ICU library this extension is linked against."""
22
+ ...
23
+
24
+ def is_locale_supported(locale: str) -> bool:
25
+ """Check if a locale is supported by ICU RBNF."""
26
+ ...
27
+
28
+ error: RBNFError
File without changes
@@ -0,0 +1,96 @@
1
+ Metadata-Version: 2.4
2
+ Name: icu_rbnf
3
+ Version: 0.2.0
4
+ Summary: Spell out numbers into words using ICU RBNF
5
+ Home-page: https://github.com/OHF-voice/icu-rbnf
6
+ Author: The Home Assistant Authors
7
+ Author-email: hello@home-assistant.io
8
+ Project-URL: Homepage, https://github.com/OHF-voice/icu-rbnf
9
+ Project-URL: Issues, https://github.com/OHF-voice/icu-rbnf/issues
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: License :: OSI Approved :: Apache Software License
13
+ Classifier: Operating System :: OS Independent
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.9
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Classifier: Programming Language :: C
21
+ Requires-Python: >=3.9
22
+ Description-Content-Type: text/markdown
23
+ Provides-Extra: dev
24
+ Requires-Dist: black; extra == "dev"
25
+ Requires-Dist: flake8; extra == "dev"
26
+ Requires-Dist: mypy; extra == "dev"
27
+ Requires-Dist: pylint; extra == "dev"
28
+ Requires-Dist: pytest; extra == "dev"
29
+ Requires-Dist: build; extra == "dev"
30
+ Dynamic: author
31
+ Dynamic: author-email
32
+ Dynamic: classifier
33
+ Dynamic: description
34
+ Dynamic: description-content-type
35
+ Dynamic: home-page
36
+ Dynamic: project-url
37
+ Dynamic: provides-extra
38
+ Dynamic: requires-python
39
+ Dynamic: summary
40
+
41
+ # icu_rbnf
42
+
43
+ A Python library for spelling out numbers into words using ICU's Rule-Based Number Format (RBNF).
44
+
45
+ ## Installation
46
+
47
+ ```bash
48
+ pip install icu_rbnf
49
+ ```
50
+
51
+ ## Usage
52
+
53
+ ```python
54
+ import icu_rbnf
55
+
56
+ # Spell out numbers in words
57
+ icu_rbnf.spellout(123, "en_US") # "one hundred twenty-three"
58
+ icu_rbnf.spellout(123.7, "en_US") # "one hundred twenty-three point seven"
59
+ icu_rbnf.spellout(123, "fr_FR") # "cent vingt-trois"
60
+
61
+ # Get ordinal form (e.g., "1st", "2nd")
62
+ icu_rbnf.ordinal(21, "en_US") # "21st"
63
+
64
+ # Get word-based ordinal (e.g., "first", "twenty-first")
65
+ icu_rbnf.spellout_ordinal(21, "en_US") # "twenty-first"
66
+
67
+ # Check if a locale is supported
68
+ icu_rbnf.is_locale_supported("en_US") # True
69
+ ```
70
+
71
+ ## API
72
+
73
+ ### `spellout(number: int | float, locale: str) -> str`
74
+
75
+ Spell out a number into words for the given locale. Supports both integers and floats.
76
+
77
+ ### `ordinal(number: int | float, locale: str) -> str`
78
+
79
+ Get the ordinal form of a number for the given locale (e.g., "1st", "2nd"). Floats are truncated to integers.
80
+
81
+ ### `spellout_ordinal(number: int | float, locale: str) -> str`
82
+
83
+ Spell out ordinal form of a number for the given locale (e.g., "first", "twenty-first"). Floats are truncated to integers. Note: Not all locales support word-based ordinals; some may fall back to numeric format.
84
+
85
+ ### `is_locale_supported(locale: str) -> bool`
86
+
87
+ Check if a locale is supported by ICU RBNF.
88
+
89
+ ### `error`
90
+
91
+ Exception class raised when ICU RBNF operations fail.
92
+
93
+ ## Requirements
94
+
95
+ - Python 3.9+
96
+ - ICU library (dynamically linked, no separate installation required)
@@ -0,0 +1,14 @@
1
+ README.md
2
+ pyproject.toml
3
+ setup.cfg
4
+ setup.py
5
+ icu_rbnf/__init__.py
6
+ icu_rbnf/__init__.pyi
7
+ icu_rbnf/py.typed
8
+ icu_rbnf.egg-info/PKG-INFO
9
+ icu_rbnf.egg-info/SOURCES.txt
10
+ icu_rbnf.egg-info/dependency_links.txt
11
+ icu_rbnf.egg-info/requires.txt
12
+ icu_rbnf.egg-info/top_level.txt
13
+ src/icu_rbnf.cpp
14
+ tests/test_icu_rbnf.py
@@ -0,0 +1,8 @@
1
+
2
+ [dev]
3
+ black
4
+ flake8
5
+ mypy
6
+ pylint
7
+ pytest
8
+ build
@@ -0,0 +1 @@
1
+ icu_rbnf
@@ -0,0 +1,3 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61.0", "wheel"]
3
+ build-backend = "setuptools.build_meta"
@@ -0,0 +1,24 @@
1
+ [bdist_wheel]
2
+ py_limited_api = cp39
3
+
4
+ [flake8]
5
+ max-line-length = 88
6
+ ignore =
7
+ E501,
8
+ W503,
9
+ E203,
10
+ D202,
11
+ W504
12
+
13
+ [isort]
14
+ multi_line_output = 3
15
+ include_trailing_comma = True
16
+ force_grid_wrap = 0
17
+ use_parentheses = True
18
+ line_length = 88
19
+ indent = " "
20
+
21
+ [egg_info]
22
+ tag_build =
23
+ tag_date = 0
24
+
@@ -0,0 +1,115 @@
1
+ #!/usr/bin/env python3
2
+ """Setup script for icu_rbnf - uses setuptools with C extension."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import os
7
+ import subprocess
8
+ import sys
9
+ from pathlib import Path
10
+
11
+ from setuptools import Extension, setup
12
+ from setuptools.command.build_ext import build_ext
13
+
14
+ # Try to find ICU using pkg-config
15
+ def get_icu_cflags():
16
+ """Get ICU compiler flags from pkg-config."""
17
+ try:
18
+ result = subprocess.run(
19
+ ["pkg-config", "--cflags", "icu-uc", "icu-i18n"],
20
+ capture_output=True,
21
+ text=True,
22
+ check=True
23
+ )
24
+ return result.stdout.strip().split()
25
+ except subprocess.CalledProcessError:
26
+ return []
27
+
28
+ def get_icu_ldflags():
29
+ """Get ICU linker flags from pkg-config."""
30
+ try:
31
+ result = subprocess.run(
32
+ ["pkg-config", "--libs", "icu-uc", "icu-i18n"],
33
+ capture_output=True,
34
+ text=True,
35
+ check=True
36
+ )
37
+ return result.stdout.strip().split()
38
+ except subprocess.CalledProcessError:
39
+ return []
40
+
41
+ class BuildExtWithICU(build_ext):
42
+ """Custom build_ext that ensures ICU is found."""
43
+
44
+ def build_extension(self, ext):
45
+ # Add ICU flags if not already present
46
+ if not ext.extra_compile_args:
47
+ ext.extra_compile_args = get_icu_cflags()
48
+ if not ext.extra_link_args:
49
+ ext.extra_link_args = get_icu_ldflags()
50
+
51
+ super().build_extension(ext)
52
+
53
+
54
+ # Define the extension module - use C++ for ICU
55
+ ext_modules = [
56
+ Extension(
57
+ "icu_rbnf._icu",
58
+ sources=["src/icu_rbnf.cpp"],
59
+ define_macros=[("PY_SSIZE_T_CLEAN", "1")],
60
+ include_dirs=[],
61
+ py_limited_api=True,
62
+ extra_compile_args=get_icu_cflags(),
63
+ extra_link_args=get_icu_ldflags(),
64
+ )
65
+ ]
66
+
67
+ # Get long description from README
68
+ readme_path = Path(__file__).parent / "README.md"
69
+ long_description = ""
70
+ if readme_path.exists():
71
+ long_description = readme_path.read_text(encoding="utf-8")
72
+
73
+ setup(
74
+ name="icu_rbnf",
75
+ version="0.2.0",
76
+ description="Spell out numbers into words using ICU RBNF",
77
+ long_description=long_description,
78
+ long_description_content_type="text/markdown",
79
+ author="The Home Assistant Authors",
80
+ author_email="hello@home-assistant.io",
81
+ url="https://github.com/OHF-voice/icu-rbnf",
82
+ project_urls={
83
+ "Homepage": "https://github.com/OHF-voice/icu-rbnf",
84
+ "Issues": "https://github.com/OHF-voice/icu-rbnf/issues",
85
+ },
86
+ packages=["icu_rbnf"],
87
+ package_dir={"icu_rbnf": "icu_rbnf"},
88
+ package_data={"icu_rbnf": ["py.typed"]},
89
+ ext_modules=ext_modules,
90
+ cmdclass={"build_ext": BuildExtWithICU},
91
+ python_requires=">=3.9",
92
+ classifiers=[
93
+ "Development Status :: 4 - Beta",
94
+ "Intended Audience :: Developers",
95
+ "License :: OSI Approved :: Apache Software License",
96
+ "Operating System :: OS Independent",
97
+ "Programming Language :: Python :: 3",
98
+ "Programming Language :: Python :: 3.9",
99
+ "Programming Language :: Python :: 3.10",
100
+ "Programming Language :: Python :: 3.11",
101
+ "Programming Language :: Python :: 3.12",
102
+ "Programming Language :: Python :: 3.13",
103
+ "Programming Language :: C",
104
+ ],
105
+ extras_require={
106
+ "dev": [
107
+ "black",
108
+ "flake8",
109
+ "mypy",
110
+ "pylint",
111
+ "pytest",
112
+ "build",
113
+ ],
114
+ },
115
+ )
@@ -0,0 +1,346 @@
1
+ #define PY_SSIZE_T_CLEAN
2
+ #include <Python.h>
3
+ #include <unicode/utypes.h>
4
+ #include <unicode/uversion.h>
5
+ #include <unicode/rbnf.h>
6
+ #include <unicode/locid.h>
7
+ #include <unicode/unistr.h>
8
+ #include <unicode/ustream.h>
9
+
10
+ static PyObject* icu_rbnf_error = NULL;
11
+
12
+ /* Helper to convert ICU UnicodeString to Python string */
13
+ static PyObject* unicode_string_to_pystring(const icu::UnicodeString* us) {
14
+ if (!us) {
15
+ Py_RETURN_NONE;
16
+ }
17
+
18
+ char* buffer = NULL;
19
+ int32_t len = 0;
20
+
21
+ /* Convert to UTF-8 */
22
+ len = us->extract(0, us->length(), NULL, 0, "UTF-8");
23
+ if (len < 0) {
24
+ PyErr_SetString(icu_rbnf_error, "Failed to get string length");
25
+ return NULL;
26
+ }
27
+
28
+ buffer = (char*)PyMem_Malloc(len + 1);
29
+ if (!buffer) {
30
+ PyErr_NoMemory();
31
+ return NULL;
32
+ }
33
+
34
+ us->extract(0, us->length(), buffer, len + 1, "UTF-8");
35
+ buffer[len] = '\0';
36
+
37
+ PyObject* result = PyUnicode_FromString(buffer);
38
+ PyMem_Free(buffer);
39
+ return result;
40
+ }
41
+
42
+ /* Helper to convert Python number to int64_t */
43
+ static int parse_number(PyObject* obj, int64_t* result) {
44
+ /* Try int first */
45
+ if (PyLong_Check(obj)) {
46
+ *result = PyLong_AsLongLong(obj);
47
+ return PyErr_Occurred() ? -1 : 0;
48
+ }
49
+
50
+ /* Try float */
51
+ if (PyFloat_Check(obj)) {
52
+ double d = PyFloat_AsDouble(obj);
53
+ if (PyErr_Occurred()) {
54
+ return -1;
55
+ }
56
+ *result = (int64_t)d;
57
+ return 0;
58
+ }
59
+
60
+ /* Try to convert via float */
61
+ PyObject* float_obj = PyNumber_Float(obj);
62
+ if (float_obj) {
63
+ double d = PyFloat_AsDouble(float_obj);
64
+ Py_DECREF(float_obj);
65
+ if (PyErr_Occurred()) {
66
+ return -1;
67
+ }
68
+ *result = (int64_t)d;
69
+ return 0;
70
+ }
71
+
72
+ PyErr_SetString(PyExc_TypeError, "argument must be a number");
73
+ return -1;
74
+ }
75
+
76
+ /* Helper to convert Python number to double */
77
+ static int parse_number_double(PyObject* obj, double* result) {
78
+ /* Try to convert via float */
79
+ PyObject* float_obj = PyNumber_Float(obj);
80
+ if (float_obj) {
81
+ *result = PyFloat_AsDouble(float_obj);
82
+ Py_DECREF(float_obj);
83
+ if (PyErr_Occurred()) {
84
+ return -1;
85
+ }
86
+ return 0;
87
+ }
88
+
89
+ PyErr_SetString(PyExc_TypeError, "argument must be a number");
90
+ return -1;
91
+ }
92
+
93
+ /* Check if a locale is supported by ICU RBNF */
94
+ static PyObject* rbnf_is_locale_supported(PyObject* self, PyObject* args) {
95
+ const char* locale_str;
96
+
97
+ if (!PyArg_ParseTuple(args, "s", &locale_str)) {
98
+ return NULL;
99
+ }
100
+
101
+ /* ICU is very permissive and falls back to default locale for invalid inputs.
102
+ * We check if the locale string is syntactically valid and if ICU can create
103
+ * a formatter with it. Since ICU always succeeds with fallback, we just verify
104
+ * that the locale string is non-empty and contains valid characters.
105
+ *
106
+ * A locale string should be in the format: language[_SCRIPT][_REGION]
107
+ * where language is 2-3 letter ISO 639 code, SCRIPT is 4 letter ISO 15924 code,
108
+ * and REGION is 2-3 letter ISO 3166 code.
109
+ */
110
+
111
+ /* Basic validation: locale string must not be empty */
112
+ if (!locale_str || locale_str[0] == '\0') {
113
+ Py_RETURN_FALSE;
114
+ }
115
+
116
+ /* Check for basic validity - should contain only alphanumeric, underscore, hyphen */
117
+ for (const char* p = locale_str; *p; ++p) {
118
+ if (!isalnum((unsigned char)*p) && *p != '_' && *p != '-') {
119
+ Py_RETURN_FALSE;
120
+ }
121
+ }
122
+
123
+ /* Try to create a spellout formatter to verify ICU can handle it */
124
+ UErrorCode status = U_ZERO_ERROR;
125
+ icu::Locale locale(locale_str);
126
+
127
+ /* Check if the locale is valid by checking its name */
128
+ icu::UnicodeString locale_name = locale.getName();
129
+ if (locale_name.isEmpty()) {
130
+ Py_RETURN_FALSE;
131
+ }
132
+
133
+ /* Try to create a formatter */
134
+ icu::RuleBasedNumberFormat* rbnf = nullptr;
135
+ rbnf = new icu::RuleBasedNumberFormat(icu::URBNF_SPELLOUT, locale, status);
136
+
137
+ if (U_FAILURE(status) || !rbnf) {
138
+ delete rbnf;
139
+ Py_RETURN_FALSE;
140
+ }
141
+
142
+ delete rbnf;
143
+ Py_RETURN_TRUE;
144
+ }
145
+
146
+ /* Spell out a number using ICU RBNF (supports both integers and floats) */
147
+ static PyObject* rbnf_spellout(PyObject* self, PyObject* args) {
148
+ PyObject* number_obj;
149
+ const char* locale_str;
150
+
151
+ if (!PyArg_ParseTuple(args, "Os", &number_obj, &locale_str)) {
152
+ return NULL;
153
+ }
154
+
155
+ /* Get the number as double (supports both int and float) */
156
+ double number;
157
+ if (parse_number_double(number_obj, &number) != 0) {
158
+ return NULL;
159
+ }
160
+
161
+ /* Create locale */
162
+ UErrorCode status = U_ZERO_ERROR;
163
+ icu::Locale locale(locale_str);
164
+
165
+ /* Create RBNF rules for spellout */
166
+ icu::RuleBasedNumberFormat* rbnf = nullptr;
167
+
168
+ /* Use the spellout format (spoken number) */
169
+ rbnf = new icu::RuleBasedNumberFormat(icu::URBNF_SPELLOUT, locale, status);
170
+
171
+ if (U_FAILURE(status) || !rbnf) {
172
+ PyErr_SetString(icu_rbnf_error, "Failed to create RBNF formatter");
173
+ delete rbnf;
174
+ return NULL;
175
+ }
176
+
177
+ /* Format the number - ICU handles both integers and floats */
178
+ icu::UnicodeString result;
179
+ rbnf->format(number, result);
180
+
181
+ delete rbnf;
182
+
183
+ return unicode_string_to_pystring(&result);
184
+ }
185
+
186
+ /* Get ordinal form of a number (e.g., "1st", "2nd") */
187
+ static PyObject* rbnf_ordinal(PyObject* self, PyObject* args) {
188
+ PyObject* number_obj;
189
+ const char* locale_str;
190
+
191
+ if (!PyArg_ParseTuple(args, "Os", &number_obj, &locale_str)) {
192
+ return NULL;
193
+ }
194
+
195
+ /* Get the number as int64_t (truncate floats) */
196
+ int64_t number;
197
+ if (parse_number(number_obj, &number) != 0) {
198
+ return NULL;
199
+ }
200
+
201
+ /* Create locale */
202
+ UErrorCode status = U_ZERO_ERROR;
203
+ icu::Locale locale(locale_str);
204
+
205
+ /* Create RBNF rules for ordinal */
206
+ icu::RuleBasedNumberFormat* rbnf = nullptr;
207
+
208
+ /* Use the ordinal format */
209
+ rbnf = new icu::RuleBasedNumberFormat(icu::URBNF_ORDINAL, locale, status);
210
+
211
+ if (U_FAILURE(status) || !rbnf) {
212
+ PyErr_SetString(icu_rbnf_error, "Failed to create RBNF formatter");
213
+ delete rbnf;
214
+ return NULL;
215
+ }
216
+
217
+ /* Format the number */
218
+ icu::UnicodeString result;
219
+ rbnf->format(number, result);
220
+
221
+ delete rbnf;
222
+
223
+ return unicode_string_to_pystring(&result);
224
+ }
225
+
226
+ /* Spell out ordinal form of a number (e.g., "first", "twenty-first") */
227
+ static PyObject* rbnf_spellout_ordinal(PyObject* self, PyObject* args) {
228
+ PyObject* number_obj;
229
+ const char* locale_str;
230
+
231
+ if (!PyArg_ParseTuple(args, "Os", &number_obj, &locale_str)) {
232
+ return NULL;
233
+ }
234
+
235
+ /* Get the number as int64_t (truncate floats) */
236
+ int64_t number;
237
+ if (parse_number(number_obj, &number) != 0) {
238
+ return NULL;
239
+ }
240
+
241
+ /* Create locale */
242
+ UErrorCode status = U_ZERO_ERROR;
243
+ icu::Locale locale(locale_str);
244
+
245
+ /* Create RBNF rules for spellout */
246
+ icu::RuleBasedNumberFormat* rbnf = nullptr;
247
+ rbnf = new icu::RuleBasedNumberFormat(icu::URBNF_SPELLOUT, locale, status);
248
+
249
+ if (U_FAILURE(status) || !rbnf) {
250
+ PyErr_SetString(icu_rbnf_error, "Failed to create RBNF formatter");
251
+ delete rbnf;
252
+ return NULL;
253
+ }
254
+
255
+ /* Try to use the %spellout-ordinal ruleset if available */
256
+ icu::UnicodeString result;
257
+ bool used_ordinal_ruleset = false;
258
+
259
+ /* Check if the ruleset exists by trying to set it */
260
+ icu::UnicodeString ruleset_name("%spellout-ordinal");
261
+ icu::UnicodeString original_default = rbnf->getDefaultRuleSetName();
262
+
263
+ status = U_ZERO_ERROR;
264
+ rbnf->setDefaultRuleSet(ruleset_name, status);
265
+
266
+ if (U_SUCCESS(status)) {
267
+ /* The ruleset exists and was set - format using it */
268
+ rbnf->format(number, result);
269
+ used_ordinal_ruleset = true;
270
+ }
271
+
272
+ /* Fallback to regular ordinal format if %spellout-ordinal not available */
273
+ if (!used_ordinal_ruleset) {
274
+ delete rbnf;
275
+ status = U_ZERO_ERROR; /* Reset status for new formatter */
276
+ rbnf = new icu::RuleBasedNumberFormat(icu::URBNF_ORDINAL, locale, status);
277
+ if (U_FAILURE(status) || !rbnf) {
278
+ PyErr_SetString(icu_rbnf_error, "Failed to create RBNF formatter");
279
+ delete rbnf;
280
+ return NULL;
281
+ }
282
+ rbnf->format(number, result);
283
+ }
284
+
285
+ delete rbnf;
286
+
287
+ return unicode_string_to_pystring(&result);
288
+ }
289
+
290
+ /* Version of the ICU the extension was built and linked against.
291
+
292
+ Worth exposing rather than leaving implicit: the spellout rules are ICU's
293
+ data, so the ICU version *is* the CLDR version, and a wheel built on an old
294
+ base image silently ships decade-old number rules. Making it readable lets a
295
+ test assert a floor. */
296
+ static PyObject* rbnf_icu_version(PyObject* self, PyObject* args) {
297
+ (void)self;
298
+ if (!PyArg_ParseTuple(args, "")) {
299
+ return NULL;
300
+ }
301
+
302
+ UVersionInfo version;
303
+ char buffer[U_MAX_VERSION_STRING_LENGTH];
304
+ u_getVersion(version);
305
+ u_versionToString(version, buffer);
306
+ return PyUnicode_FromString(buffer);
307
+ }
308
+
309
+ /* Module methods */
310
+ static PyMethodDef IcuRbnfMethods[] = {
311
+ {"is_locale_supported", rbnf_is_locale_supported, METH_VARARGS, "Check if a locale is supported by ICU RBNF"},
312
+ {"spellout", rbnf_spellout, METH_VARARGS, "Spell out a number in words"},
313
+ {"icu_version", rbnf_icu_version, METH_VARARGS, "ICU library version this extension is linked against"},
314
+ {"ordinal", rbnf_ordinal, METH_VARARGS, "Get ordinal form of a number (e.g., '1st', '2nd')"},
315
+ {"spellout_ordinal", rbnf_spellout_ordinal, METH_VARARGS, "Spell out ordinal form of a number (e.g., 'first', 'twenty-first')"},
316
+ {NULL, NULL, 0, NULL}
317
+ };
318
+
319
+ /* Module definition */
320
+ static struct PyModuleDef icu_rbnf_module = {
321
+ PyModuleDef_HEAD_INIT,
322
+ "_icu",
323
+ "ICU RBNF extension module",
324
+ -1,
325
+ IcuRbnfMethods,
326
+ NULL,
327
+ NULL,
328
+ NULL,
329
+ NULL
330
+ };
331
+
332
+ /* Module initialization */
333
+ PyMODINIT_FUNC PyInit__icu(void) {
334
+ PyObject* m = PyModule_Create(&icu_rbnf_module);
335
+ if (!m) {
336
+ return NULL;
337
+ }
338
+
339
+ /* Create exception class */
340
+ icu_rbnf_error = PyErr_NewException("icu_rbnf.error", NULL, NULL);
341
+ if (icu_rbnf_error) {
342
+ PyModule_AddObject(m, "error", icu_rbnf_error);
343
+ }
344
+
345
+ return m;
346
+ }
@@ -0,0 +1,248 @@
1
+ """Tests for icu_rbnf library."""
2
+
3
+ import icu_rbnf
4
+
5
+
6
+ class TestSpellout:
7
+ """Tests for the spellout function."""
8
+
9
+ def test_english_basic(self):
10
+ """Test basic number spellout in English."""
11
+ assert icu_rbnf.spellout(0, "en_US") == "zero"
12
+ assert icu_rbnf.spellout(1, "en_US") == "one"
13
+ assert icu_rbnf.spellout(12, "en_US") == "twelve"
14
+ assert icu_rbnf.spellout(20, "en_US") == "twenty"
15
+ assert icu_rbnf.spellout(42, "en_US") == "forty-two"
16
+ assert icu_rbnf.spellout(100, "en_US") == "one hundred"
17
+ assert icu_rbnf.spellout(123, "en_US") == "one hundred twenty-three"
18
+ assert icu_rbnf.spellout(999, "en_US") == "nine hundred ninety-nine"
19
+
20
+ def test_french_basic(self):
21
+ """Test basic number spellout in French."""
22
+ assert icu_rbnf.spellout(0, "fr_FR") == "zéro"
23
+ assert icu_rbnf.spellout(1, "fr_FR") == "un"
24
+ assert icu_rbnf.spellout(11, "fr_FR") == "onze"
25
+ assert icu_rbnf.spellout(21, "fr_FR") == "vingt-et-un"
26
+ assert icu_rbnf.spellout(100, "fr_FR") == "cent"
27
+ assert icu_rbnf.spellout(123, "fr_FR") == "cent vingt-trois"
28
+
29
+ def test_german_basic(self):
30
+ """Test basic number spellout in German."""
31
+ assert icu_rbnf.spellout(0, "de_DE") == "null"
32
+ assert icu_rbnf.spellout(1, "de_DE") == "eins"
33
+ assert icu_rbnf.spellout(12, "de_DE") == "zwölf"
34
+ # German uses soft hyphen in some compound words (U+00AD)
35
+ result = icu_rbnf.spellout(21, "de_DE")
36
+ assert "ein" in result and "zwanzig" in result
37
+ # ICU may use soft hyphen in compound words
38
+ result_100 = icu_rbnf.spellout(100, "de_DE")
39
+ assert "ein" in result_100 and "hundert" in result_100
40
+
41
+ def test_large_numbers(self):
42
+ """Test spellout with larger numbers."""
43
+ assert icu_rbnf.spellout(1000, "en_US") == "one thousand"
44
+ assert (
45
+ icu_rbnf.spellout(1234, "en_US") == "one thousand two hundred thirty-four"
46
+ )
47
+ assert icu_rbnf.spellout(1000000, "en_US") == "one million"
48
+
49
+ def test_example_from_spec(self):
50
+ """Test the example from the specification."""
51
+ assert icu_rbnf.spellout(123, "en_US") == "one hundred twenty-three"
52
+ assert icu_rbnf.spellout(123, "fr_FR") == "cent vingt-trois"
53
+
54
+
55
+ class TestOrdinal:
56
+ """Tests for the ordinal function."""
57
+
58
+ def test_english_basic(self):
59
+ """Test basic ordinal forms in English."""
60
+ assert icu_rbnf.ordinal(1, "en_US") == "1st"
61
+ assert icu_rbnf.ordinal(2, "en_US") == "2nd"
62
+ assert icu_rbnf.ordinal(3, "en_US") == "3rd"
63
+ assert icu_rbnf.ordinal(4, "en_US") == "4th"
64
+ assert icu_rbnf.ordinal(10, "en_US") == "10th"
65
+ assert icu_rbnf.ordinal(11, "en_US") == "11th"
66
+ assert icu_rbnf.ordinal(21, "en_US") == "21st"
67
+ assert icu_rbnf.ordinal(22, "en_US") == "22nd"
68
+ assert icu_rbnf.ordinal(100, "en_US") == "100th"
69
+ assert icu_rbnf.ordinal(101, "en_US") == "101st"
70
+
71
+ def test_french_basic(self):
72
+ """Test basic ordinal forms in French."""
73
+ assert icu_rbnf.ordinal(1, "fr_FR") == "1er"
74
+ assert icu_rbnf.ordinal(2, "fr_FR") == "2e"
75
+ assert icu_rbnf.ordinal(3, "fr_FR") == "3e"
76
+ assert icu_rbnf.ordinal(10, "fr_FR") == "10e"
77
+ assert icu_rbnf.ordinal(20, "fr_FR") == "20e"
78
+
79
+ def test_german_basic(self):
80
+ """Test basic ordinal forms in German."""
81
+ assert icu_rbnf.ordinal(1, "de_DE") == "1."
82
+ assert icu_rbnf.ordinal(2, "de_DE") == "2."
83
+ assert icu_rbnf.ordinal(3, "de_DE") == "3."
84
+ assert icu_rbnf.ordinal(10, "de_DE") == "10."
85
+
86
+ def test_example_from_spec(self):
87
+ """Test the example from the specification."""
88
+ assert icu_rbnf.ordinal(21, "en_US") == "21st"
89
+
90
+
91
+ class TestEdgeCases:
92
+ """Tests for edge cases."""
93
+
94
+ def test_negative_numbers(self):
95
+ """Test spellout with negative numbers."""
96
+ assert icu_rbnf.spellout(-1, "en_US") == "minus one"
97
+ assert icu_rbnf.spellout(-42, "en_US") == "minus forty-two"
98
+
99
+ def test_float_numbers(self):
100
+ """Test spellout with float numbers (properly formatted)."""
101
+ assert (
102
+ icu_rbnf.spellout(123.7, "en_US") == "one hundred twenty-three point seven"
103
+ )
104
+ assert icu_rbnf.spellout(42.0, "en_US") == "forty-two"
105
+
106
+ def test_invalid_locale(self):
107
+ """Test behavior with invalid locale."""
108
+ # ICU may not raise an error for invalid locale, just use default
109
+ # This test verifies the error exception is available
110
+ assert hasattr(icu_rbnf, "error")
111
+
112
+ def test_very_large_number(self):
113
+ """Test with a very large number."""
114
+ result = icu_rbnf.spellout(999999999, "en_US")
115
+ assert "nine hundred ninety-nine million" in result
116
+
117
+
118
+ class TestSpelloutOrdinal:
119
+ """Tests for the spellout_ordinal function."""
120
+
121
+ def test_english_basic(self):
122
+ """Test basic word-based ordinals in English."""
123
+ assert icu_rbnf.spellout_ordinal(1, "en_US") == "first"
124
+ assert icu_rbnf.spellout_ordinal(2, "en_US") == "second"
125
+ assert icu_rbnf.spellout_ordinal(3, "en_US") == "third"
126
+ assert icu_rbnf.spellout_ordinal(4, "en_US") == "fourth"
127
+ assert icu_rbnf.spellout_ordinal(10, "en_US") == "tenth"
128
+ assert icu_rbnf.spellout_ordinal(11, "en_US") == "eleventh"
129
+ assert icu_rbnf.spellout_ordinal(21, "en_US") == "twenty-first"
130
+ assert icu_rbnf.spellout_ordinal(22, "en_US") == "twenty-second"
131
+ assert icu_rbnf.spellout_ordinal(100, "en_US") == "one hundredth"
132
+ assert icu_rbnf.spellout_ordinal(123, "en_US") == "one hundred twenty-third"
133
+
134
+ def test_german_basic(self):
135
+ """Test basic word-based ordinals in German."""
136
+ assert icu_rbnf.spellout_ordinal(1, "de_DE") == "erste"
137
+ assert icu_rbnf.spellout_ordinal(2, "de_DE") == "zweite"
138
+ assert icu_rbnf.spellout_ordinal(3, "de_DE") == "dritte"
139
+ assert icu_rbnf.spellout_ordinal(10, "de_DE") == "zehnte"
140
+
141
+ def test_fallback_behavior(self):
142
+ """Test that spellout_ordinal falls back gracefully for locales without %spellout-ordinal."""
143
+ # French doesn't have %spellout-ordinal, so it should fall back to ordinal format
144
+ # This test verifies the fallback mechanism works
145
+ result = icu_rbnf.spellout_ordinal(1, "fr_FR")
146
+ # Should return ordinal format (1er) since word-based isn't available
147
+ assert "1" in result and "er" in result
148
+
149
+
150
+ class TestIsLocaleSupported:
151
+ """Tests for the is_locale_supported function."""
152
+
153
+ def test_valid_locales(self):
154
+ """Test that valid locales are supported."""
155
+ assert icu_rbnf.is_locale_supported("en_US") is True
156
+ assert icu_rbnf.is_locale_supported("fr_FR") is True
157
+ assert icu_rbnf.is_locale_supported("de_DE") is True
158
+ assert icu_rbnf.is_locale_supported("es_ES") is True
159
+ assert icu_rbnf.is_locale_supported("en_GB") is True
160
+
161
+ def test_invalid_locales(self):
162
+ """Test that empty strings are not supported."""
163
+ assert icu_rbnf.is_locale_supported("") is False
164
+
165
+ def test_special_characters(self):
166
+ """Test that locales with invalid characters are rejected."""
167
+ assert icu_rbnf.is_locale_supported("en@US") is False
168
+ assert icu_rbnf.is_locale_supported("en US") is False
169
+
170
+
171
+ class TestModuleAPI:
172
+ """Tests for module-level API."""
173
+
174
+ def test_error_exception_exists(self):
175
+ """Test that error exception is exposed at module level."""
176
+ assert hasattr(icu_rbnf, "error")
177
+ assert isinstance(icu_rbnf.error, type)
178
+
179
+ def test_version_exists(self):
180
+ """Test that version is exposed."""
181
+ assert hasattr(icu_rbnf, "__version__")
182
+ # Shape, not a literal: pinning the value here means every release
183
+ # bump fails a test that is not about the release.
184
+ parts = icu_rbnf.__version__.split(".")
185
+ assert len(parts) >= 2
186
+ assert all(p.isdigit() for p in parts[:2])
187
+
188
+ def test_spellout_ordinal_exists(self):
189
+ """Test that spellout_ordinal function is exposed."""
190
+ assert hasattr(icu_rbnf, "spellout_ordinal")
191
+ assert callable(icu_rbnf.spellout_ordinal)
192
+
193
+ def test_is_locale_supported_exists(self):
194
+ """Test that is_locale_supported function is exposed."""
195
+ assert hasattr(icu_rbnf, "is_locale_supported")
196
+ assert callable(icu_rbnf.is_locale_supported)
197
+
198
+
199
+ class TestIcuVersion:
200
+ """The ICU version is the CLDR version, so it is worth asserting."""
201
+
202
+ def test_reports_a_version(self):
203
+ """icu_version() returns something like '78.3'."""
204
+ version = icu_rbnf.icu_version()
205
+ assert version
206
+ assert version[0].isdigit()
207
+
208
+ def test_is_recent_enough(self):
209
+ """Guard against a build picking up an ancient system ICU.
210
+
211
+ ICU 60 -- AlmaLinux 8's, and what these wheels shipped before
212
+ script/build-icu existed -- carries CLDR 32 from 2017. It has no rules
213
+ at all for several locales and silently falls back to English for them,
214
+ which is indistinguishable from working. ICU 77 is also the first
215
+ release whose RBNF parser understands CLDR's `[a >>|b]` syntax.
216
+ """
217
+ major = int(icu_rbnf.icu_version().split(".", maxsplit=1)[0])
218
+ assert major >= 77, f"ICU {major} is too old; see script/build-icu"
219
+
220
+
221
+ class TestLocalesMissingFromOldIcu:
222
+ """Locales that ICU 60 did not have, and silently spelled in English.
223
+
224
+ Each of these returned the English word before the ICU upgrade, so they
225
+ double as a check that the build did not quietly regress to a system ICU.
226
+ """
227
+
228
+ def test_swahili(self):
229
+ assert icu_rbnf.spellout(1, "sw") == "moja"
230
+ assert icu_rbnf.spellout(2, "sw") == "mbili"
231
+
232
+ def test_luxembourgish(self):
233
+ assert icu_rbnf.spellout(0, "lb") == "null"
234
+ assert icu_rbnf.spellout(1, "lb") == "eent"
235
+
236
+ def test_kazakh(self):
237
+ assert icu_rbnf.spellout(0, "kk") == "нөл"
238
+ assert icu_rbnf.spellout(1, "kk") == "бір"
239
+
240
+ def test_nepali(self):
241
+ assert icu_rbnf.spellout(1, "ne") == "एक"
242
+
243
+ def test_sundanese(self):
244
+ assert icu_rbnf.spellout(0, "su") == "nol"
245
+ assert icu_rbnf.spellout(1, "su") == "hiji"
246
+
247
+ def test_quechua(self):
248
+ assert icu_rbnf.spellout(1, "qu") == "huk"