numbr 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- numbr/__init__.py +37 -0
- numbr/engine.py +1180 -0
- numbr-1.0.0.dist-info/METADATA +110 -0
- numbr-1.0.0.dist-info/RECORD +6 -0
- numbr-1.0.0.dist-info/WHEEL +5 -0
- numbr-1.0.0.dist-info/top_level.txt +1 -0
numbr/__init__.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
from . import engine as __engine
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
# "replaceNumericValue",
|
|
8
|
+
"wordsToInt",
|
|
9
|
+
"ordinalSuffix",
|
|
10
|
+
"intToWords",
|
|
11
|
+
"intToOrdinalWords",
|
|
12
|
+
"stripOrdinalSuffix",
|
|
13
|
+
"ordinalWordsToInt",
|
|
14
|
+
"stringToInt",
|
|
15
|
+
"extractNumericValue",
|
|
16
|
+
"romanToWords",
|
|
17
|
+
"romanToInt",
|
|
18
|
+
"formatDecimal",
|
|
19
|
+
"insertSep",
|
|
20
|
+
]
|
|
21
|
+
|
|
22
|
+
# Reference functions using __engine alias
|
|
23
|
+
# replaceNumericValue = __engine.replaceNumericValue
|
|
24
|
+
wordsToInt = __engine.wordsToInt
|
|
25
|
+
ordinalSuffix = __engine.ordinalSuffix
|
|
26
|
+
intToWords = __engine.intToWords
|
|
27
|
+
intToOrdinalWords = __engine.intToOrdinalWords
|
|
28
|
+
stripOrdinalSuffix = __engine.stripOrdinalSuffix
|
|
29
|
+
ordinalWordsToInt = __engine.ordinalWordsToInt
|
|
30
|
+
stringToInt = __engine.stringToInt
|
|
31
|
+
extractNumericValue = __engine.extractNumericValue
|
|
32
|
+
romanToWords = __engine.romanToWords
|
|
33
|
+
romanToInt = __engine.romanToInt
|
|
34
|
+
formatDecimal = __engine.formatDecimal
|
|
35
|
+
insertSep = __engine.insertSep
|
|
36
|
+
|
|
37
|
+
del engine
|
numbr/engine.py
ADDED
|
@@ -0,0 +1,1180 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
|
|
3
|
+
#
|
|
4
|
+
# Understanding Number Terminology
|
|
5
|
+
# ─────────────────────────────────
|
|
6
|
+
# ┍━━━━━━━━━┯━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┯━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┑
|
|
7
|
+
# │ Example │ Type │ Description │
|
|
8
|
+
# ┝━━━━━━━━━┿━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┿━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┥
|
|
9
|
+
# │ four │ Cardinal number (word form) │ This is the written-out version of the number 4. │
|
|
10
|
+
# │ 4 │ Cardinal numeral (digit form) │ This is the numeric symbol representing "four." │
|
|
11
|
+
# │ fourth │ Ordinal number (word form) │ This is the written-out version of "4th," used to describe position. │
|
|
12
|
+
# │ 4th │ Ordinal numeral (digit form) │ This is the numeric way of writing an ordinal number. │
|
|
13
|
+
# │ IV │ Roman numeral │ This represents the number "4" in the Roman numeral system. │
|
|
14
|
+
# ┕━━━━━━━━━┷━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┷━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┙
|
|
15
|
+
#
|
|
16
|
+
# This module is designed to help developers work with numbers in natural language processing (NLP), data extraction,
|
|
17
|
+
# and automated text conversion.
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
import re
|
|
22
|
+
import inspect
|
|
23
|
+
from typing import Literal, Union, Optional
|
|
24
|
+
from functools import reduce
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
## Module-Level Constants & Dictionaries
|
|
29
|
+
##━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
|
|
30
|
+
|
|
31
|
+
# Matches spelled-out numbers, including cardinal and ordinal forms.
|
|
32
|
+
_NUM_ORDINAL_WRDS_RE = (
|
|
33
|
+
r'zero|one|two|three|four|five|six|seven|eight|nine|ten|'
|
|
34
|
+
r'eleven|twelve|thirteen|fourteen|fifteen|sixteen|seventeen|eighteen|nineteen|'
|
|
35
|
+
r'twenty|thirty|forty|fifty|sixty|seventy|eighty|ninety|'
|
|
36
|
+
r'hundred|thousand|million|billion|trillion|quadrillion|quintillion|'
|
|
37
|
+
r'first|second|third|fourth|fifth|sixth|seventh|eighth|ninth|tenth|'
|
|
38
|
+
r'eleventh|twelfth|thirteenth|fourteenth|fifteenth|sixteenth|seventeenth|'
|
|
39
|
+
r'eighteenth|nineteenth|twentieth|thirtieth|fortieth|fiftieth|sixtieth|'
|
|
40
|
+
r'seventieth|eightieth|ninetieth|hundredth|thousandth|millionth|billionth|'
|
|
41
|
+
r'trillionth|quadrillionth|quintillionth'
|
|
42
|
+
)
|
|
43
|
+
# # Matches entire spelled-out numbers, allowing hyphenated and space-separated formats.
|
|
44
|
+
# # Also accommodates "and" usage within numbers (e.g., "one hundred and twenty").
|
|
45
|
+
_NUM_WORDS_RE = (
|
|
46
|
+
r'\b(?:'
|
|
47
|
+
+ _NUM_ORDINAL_WRDS_RE
|
|
48
|
+
+ r')(?:[-\s]+(?:and[-\s]+)?(?:'
|
|
49
|
+
+ _NUM_ORDINAL_WRDS_RE
|
|
50
|
+
+ r'))*\b'
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
# Basic unit digits (1-9) as words.
|
|
54
|
+
_UNIT_DIGITS_WORDS = ["one", "two", "three", "four", "five", "six", "seven", "eight", "nine"]
|
|
55
|
+
|
|
56
|
+
# Multiples of ten (10-90) as words.
|
|
57
|
+
_TENS_MULTIPLES_WORDS = ["ten", "twenty", "thirty", "forty", "fifty", "sixty", "seventy", "eighty", "ninety"]
|
|
58
|
+
|
|
59
|
+
# Special cases for numbers between 11 and 19.
|
|
60
|
+
_TEEN_NUMERALS_WORDS = ["eleven", "twelve", "thirteen", "fourteen", "fifteen", "sixteen", "seventeen", "eighteen", "nineteen"]
|
|
61
|
+
|
|
62
|
+
# Maps ordinal words (e.g., "first", "second", "twentieth") to their corresponding numeric values.
|
|
63
|
+
_ORDINAL_MAPPING = {
|
|
64
|
+
"first": 1, "second": 2, "third": 3, "fourth": 4, "fifth": 5,
|
|
65
|
+
"sixth": 6, "seventh": 7, "eighth": 8, "ninth": 9, "tenth": 10,
|
|
66
|
+
"eleventh": 11, "twelfth": 12, "thirteenth": 13, "fourteenth": 14,
|
|
67
|
+
"fifteenth": 15, "sixteenth": 16, "seventeenth": 17, "eighteenth": 18,
|
|
68
|
+
"nineteenth": 19, "twentieth": 20, "thirtieth": 30, "fortieth": 40,
|
|
69
|
+
"fiftieth": 50, "sixtieth": 60, "seventieth": 70, "eightieth": 80,
|
|
70
|
+
"ninetieth": 90, "hundredth": 100, "thousandth": 1000,
|
|
71
|
+
"millionth": 1000000, "billionth": 1000000000,
|
|
72
|
+
"trillionth": 1000000000000,
|
|
73
|
+
"quadrillionth": 1000000000000000,
|
|
74
|
+
"quintillionth": 1000000000000000000,
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
# Dictionary that maps spelled-out ordinal words to their corresponding cardinal form and ordinal suffix.
|
|
78
|
+
_WORD_BASED_PATTERNS_RE = {
|
|
79
|
+
r'first$': ('one', 'st'),
|
|
80
|
+
r'second$': ('two', 'nd'),
|
|
81
|
+
r'third$': ('three', 'rd'),
|
|
82
|
+
r'fourth$': ('four', 'th'),
|
|
83
|
+
r'fifth$': ('five', 'th'),
|
|
84
|
+
r'sixth$': ('six', 'th'),
|
|
85
|
+
r'seventh$': ('seven', 'th'),
|
|
86
|
+
r'eighth$': ('eight', 'th'),
|
|
87
|
+
r'ninth$': ('nine', 'th'),
|
|
88
|
+
r'tenth$': ('ten', 'th'),
|
|
89
|
+
r'eleventh$': ('eleven', 'th'),
|
|
90
|
+
r'twelfth$': ('twelve', 'th'),
|
|
91
|
+
r'thirteenth$': ('thirteen', 'th'),
|
|
92
|
+
r'fourteenth$': ('fourteen', 'th'),
|
|
93
|
+
r'fifteenth$': ('fifteen', 'th'),
|
|
94
|
+
r'sixteenth$': ('sixteen', 'th'),
|
|
95
|
+
r'seventeenth$': ('seventeen', 'th'),
|
|
96
|
+
r'eighteenth$': ('eighteen', 'th'),
|
|
97
|
+
r'nineteenth$': ('nineteen', 'th'),
|
|
98
|
+
r'twentieth$': ('twenty', 'th'),
|
|
99
|
+
r'thirtieth$': ('thirty', 'th'),
|
|
100
|
+
r'fortieth$': ('forty', 'th'),
|
|
101
|
+
r'fiftieth$': ('fifty', 'th'),
|
|
102
|
+
r'sixtieth$': ('sixty', 'th'),
|
|
103
|
+
r'seventieth$': ('seventy', 'th'),
|
|
104
|
+
r'eightieth$': ('eighty', 'th'),
|
|
105
|
+
r'ninetieth$': ('ninety', 'th'),
|
|
106
|
+
r'hundredth$': ('hundred', 'th'),
|
|
107
|
+
r'thousandth$': ('thousand', 'th'),
|
|
108
|
+
r'millionth$': ('million', 'th'),
|
|
109
|
+
r'billionth$': ('billion', 'th'),
|
|
110
|
+
r'trillionth$': ('trillion', 'th'),
|
|
111
|
+
r'quadrillionth$': ('quadrillion', 'th'),
|
|
112
|
+
r'quintillionth$': ('quintillion', 'th'),
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
# Mapping for number multipliers used to scale values.
|
|
116
|
+
_MULTIPLIERS = {
|
|
117
|
+
"hundred": 100,
|
|
118
|
+
"thousand": 1000,
|
|
119
|
+
"million": 1000000,
|
|
120
|
+
"billion": 1000000000,
|
|
121
|
+
"trillion": 1000000000000,
|
|
122
|
+
"quadrillion": 1000000000000000,
|
|
123
|
+
"quintillion": 1000000000000000000,
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
# Maps individual Roman numeral symbols to their corresponding integer values based on the standard numeral system.
|
|
127
|
+
_ROMAN_NUMERAL_MAPPING = {
|
|
128
|
+
'I': 1, 'V': 5, 'X': 10, 'L': 50,
|
|
129
|
+
'C': 100, 'D': 500, 'M': 1000
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
# This regex pattern ensures strict validation of Roman numerals.
|
|
133
|
+
# It follows the standard Roman numeral rules and supports numbers from 1 (I) to 3999 (MMMCMXCIX).
|
|
134
|
+
# The pattern is divided into four groups:
|
|
135
|
+
# - (M{0,3}) → Matches 0 to 3 occurrences of 'M' (1000s place).
|
|
136
|
+
# - (CM|CD|D?C{0,3}) → Matches 900 (CM), 400 (CD), or up to 3 'C' after optional 'D' (100s place).
|
|
137
|
+
# - (XC|XL|L?X{0,3}) → Matches 90 (XC), 40 (XL), or up to 3 'X' after optional 'L' (10s place).
|
|
138
|
+
# - (IX|IV|V?I{0,3}) → Matches 9 (IX), 4 (IV), or up to 3 'I' after optional 'V' (1s place).
|
|
139
|
+
_ROMAN_NUMERAL_RE = r"^(M{0,3})(CM|CD|D?C{0,3})(XC|XL|L?X{0,3})(IX|IV|V?I{0,3})$"
|
|
140
|
+
|
|
141
|
+
# List of common currency symbols from around the world
|
|
142
|
+
_CURRENCY_SYMBOLS = [
|
|
143
|
+
r"$", # US Dollar
|
|
144
|
+
r"€", # Euro
|
|
145
|
+
r"£", # British Pound Sterling
|
|
146
|
+
r"¥", # Japanese Yen / Chinese Yuan
|
|
147
|
+
r"₹", # Indian Rupee
|
|
148
|
+
r"₩", # South Korean Won
|
|
149
|
+
r"₽", # Russian Ruble
|
|
150
|
+
r"R$", # Brazilian Real
|
|
151
|
+
r"₺", # Turkish Lira
|
|
152
|
+
r"฿", # Thai Baht
|
|
153
|
+
r"₫", # Vietnamese Dong
|
|
154
|
+
r"₱", # Philippine Peso
|
|
155
|
+
r"₴", # Ukrainian Hryvnia
|
|
156
|
+
r"₸", # Kazakhstani Tenge
|
|
157
|
+
r"֏", # Armenian Dram
|
|
158
|
+
r"₦", # Nigerian Naira
|
|
159
|
+
r"₵", # Ghanaian Cedi
|
|
160
|
+
r"Br", # Belarusian Ruble / Ethiopian Birr
|
|
161
|
+
r"₾", # Georgian Lari
|
|
162
|
+
r"₪", # Israeli Shekel
|
|
163
|
+
r"R", # South African Rand
|
|
164
|
+
r"HK$", # Hong Kong Dollar
|
|
165
|
+
r"S$", # Singapore Dollar
|
|
166
|
+
r"RM", # Malaysian Ringgit
|
|
167
|
+
r"Rp", # Indonesian Rupiah
|
|
168
|
+
r"Kč", # Czech Koruna
|
|
169
|
+
r"zł", # Polish Zloty
|
|
170
|
+
r"kr", # Scandinavian Krone (Denmark, Norway, Sweden)
|
|
171
|
+
r"Ft", # Hungarian Forint
|
|
172
|
+
r"lei", # Romanian Leu
|
|
173
|
+
r"лв", # Bulgarian Lev
|
|
174
|
+
r"дин", # Serbian Dinar
|
|
175
|
+
r"kn", # Croatian Kuna
|
|
176
|
+
r"ден", # Macedonian Denar
|
|
177
|
+
r"L", # Albanian Lek
|
|
178
|
+
r"Br", # Repeated for Belarusian Ruble / Ethiopian Birr
|
|
179
|
+
r"S/.", # Peruvian Sol
|
|
180
|
+
r"CHF" # Swiss Franc
|
|
181
|
+
]
|
|
182
|
+
|
|
183
|
+
## Helper Functions
|
|
184
|
+
##━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
# String Parsing & Tokenization
|
|
188
|
+
#────────────────────────────────────────────────────────────────────────────
|
|
189
|
+
def __parseNumericToken(s: str, first_only: bool = True, wrap_single: bool = False):
|
|
190
|
+
"""
|
|
191
|
+
Finds spelled-out or digit-based tokens in a string and returns
|
|
192
|
+
either the first match or all.
|
|
193
|
+
|
|
194
|
+
This function searches for numeric tokens in two ways:
|
|
195
|
+
1. Digit-based (e.g., '42', '1,234', '2.50')
|
|
196
|
+
2. Spelled-out (e.g., 'twenty-five', 'third')
|
|
197
|
+
"""
|
|
198
|
+
def _tokenize_digit(text: str, first_only: bool = True):
|
|
199
|
+
"""
|
|
200
|
+
Finds digit-based tokens in a string, including optional decimals, and commas.
|
|
201
|
+
"""
|
|
202
|
+
pattern = r'-?\d+(?:,\d{3})*(?:\.\d+)?'
|
|
203
|
+
|
|
204
|
+
matches = list(re.finditer(pattern, text))
|
|
205
|
+
if not matches:
|
|
206
|
+
return None if first_only else []
|
|
207
|
+
|
|
208
|
+
# Convert each match to a string, removing commas
|
|
209
|
+
found = [re.sub(r',', '', m.group(0)) for m in matches]
|
|
210
|
+
|
|
211
|
+
if first_only:
|
|
212
|
+
# Return a single-element list
|
|
213
|
+
return [found[0]]
|
|
214
|
+
else:
|
|
215
|
+
return found
|
|
216
|
+
|
|
217
|
+
def _tokenize_alpha(text: str, first_only: bool = True):
|
|
218
|
+
"""
|
|
219
|
+
Finds spelled-out numbers or ordinals in a string, using _NUM_WORDS_RE.
|
|
220
|
+
"""
|
|
221
|
+
matches = list(re.finditer(_NUM_WORDS_RE, text, re.IGNORECASE))
|
|
222
|
+
if not matches:
|
|
223
|
+
return None if first_only else []
|
|
224
|
+
|
|
225
|
+
# Extract each spelled-out token
|
|
226
|
+
found = [text[m.start():m.end()] for m in matches]
|
|
227
|
+
|
|
228
|
+
if first_only:
|
|
229
|
+
return [found[0]]
|
|
230
|
+
else:
|
|
231
|
+
return found
|
|
232
|
+
|
|
233
|
+
if not s:
|
|
234
|
+
return None if first_only else []
|
|
235
|
+
|
|
236
|
+
# Get all digit-based tokens
|
|
237
|
+
digit_tokens = _tokenize_digit(s, first_only=False) # get all
|
|
238
|
+
# Get all spelled-out tokens
|
|
239
|
+
alpha_tokens = _tokenize_alpha(s, first_only=False) # get all
|
|
240
|
+
|
|
241
|
+
# Combine results
|
|
242
|
+
combined = []
|
|
243
|
+
if digit_tokens:
|
|
244
|
+
combined += digit_tokens
|
|
245
|
+
if alpha_tokens:
|
|
246
|
+
combined += alpha_tokens
|
|
247
|
+
|
|
248
|
+
if not combined:
|
|
249
|
+
return None if first_only else []
|
|
250
|
+
|
|
251
|
+
# If user wants only the first match, return the first item
|
|
252
|
+
if first_only:
|
|
253
|
+
result = combined[0]
|
|
254
|
+
return [result] if wrap_single else result # Wrap in list if requested
|
|
255
|
+
else:
|
|
256
|
+
return combined
|
|
257
|
+
|
|
258
|
+
def __parseRomanNumeral(s: str):
|
|
259
|
+
"""
|
|
260
|
+
Parses a Roman numeral string and converts it into an integer.
|
|
261
|
+
|
|
262
|
+
This function ensures the Roman numeral follows valid ordering rules and
|
|
263
|
+
correctly applies subtractive notation.
|
|
264
|
+
"""
|
|
265
|
+
if not s or not re.match(_ROMAN_NUMERAL_RE, s):
|
|
266
|
+
return None
|
|
267
|
+
|
|
268
|
+
num, prev = 0, 0
|
|
269
|
+
|
|
270
|
+
# Convert from right to left
|
|
271
|
+
for char in reversed(s):
|
|
272
|
+
curr = _ROMAN_NUMERAL_MAPPING[char]
|
|
273
|
+
num = num - curr if curr < prev else num + curr
|
|
274
|
+
prev = curr
|
|
275
|
+
|
|
276
|
+
return num if num > 0 else None # Ensure non-zero positive value
|
|
277
|
+
|
|
278
|
+
def __validate_numstr(n: Union[int, float, str], clean: bool = False) -> Optional[str]:
|
|
279
|
+
"""
|
|
280
|
+
Validates whether the input contains a valid numeric value after removing
|
|
281
|
+
currency symbols and commas.
|
|
282
|
+
|
|
283
|
+
This function converts the input to a string, removes known currency symbols
|
|
284
|
+
and commas, and checks if the remaining content is a valid numeric value.
|
|
285
|
+
"""
|
|
286
|
+
numstr = str(n)
|
|
287
|
+
clean_numstr = reduce(lambda s, sign: s.replace(sign, ""), _CURRENCY_SYMBOLS, numstr).replace(",", "")
|
|
288
|
+
clean_numstr = clean_numstr.replace("+", "").replace("-", "")
|
|
289
|
+
# Check if the string is a valid number
|
|
290
|
+
try:
|
|
291
|
+
float(clean_numstr) # Check for validity
|
|
292
|
+
except ValueError:
|
|
293
|
+
return None
|
|
294
|
+
return clean_numstr if clean else n
|
|
295
|
+
|
|
296
|
+
# Boolean & Utility Functions
|
|
297
|
+
#────────────────────────────────────────────────────────────────────────────
|
|
298
|
+
def __switch(x):
|
|
299
|
+
"""
|
|
300
|
+
Toggle a boolean value.
|
|
301
|
+
|
|
302
|
+
Parameters:
|
|
303
|
+
──────────────────────────
|
|
304
|
+
x (bool): The boolean value to switch.
|
|
305
|
+
|
|
306
|
+
Returns:
|
|
307
|
+
──────────────────────────
|
|
308
|
+
bool: The opposite of the input boolean value.
|
|
309
|
+
"""
|
|
310
|
+
return not x
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
## General Formatting & Numeric Processing
|
|
316
|
+
##━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
|
|
317
|
+
|
|
318
|
+
# These are intermediate utility functions for number formatting
|
|
319
|
+
#────────────────────────────────────────────────────────────────────────────
|
|
320
|
+
def insertSep(n: Union[int, float, str], sep: str = ",") -> Optional[str]:
|
|
321
|
+
"""
|
|
322
|
+
Formats a number by inserting a specified thousands separator while preserving
|
|
323
|
+
any non-numeric characters (e.g., currency symbols, quotes).
|
|
324
|
+
|
|
325
|
+
This function extracts the numeric portion of the input, formats it with
|
|
326
|
+
the specified thousands separator, and then restores the original non-numeric
|
|
327
|
+
characters at the beginning and end.
|
|
328
|
+
|
|
329
|
+
Parameters:
|
|
330
|
+
──────────────────────────
|
|
331
|
+
n (str, int, or float): The input number, which may contain non-numeric characters.
|
|
332
|
+
sep (str, optional): The separator to use for thousands grouping. Default is a comma (",").
|
|
333
|
+
|
|
334
|
+
Returns:
|
|
335
|
+
──────────────────────────
|
|
336
|
+
str: The formatted number with thousands separators, retaining any original non-numeric characters.
|
|
337
|
+
None: If the input does not contain a valid number.
|
|
338
|
+
"""
|
|
339
|
+
num_str = str(n) # Ensure the input is a string
|
|
340
|
+
num_str = " ".join(num_str.split())
|
|
341
|
+
|
|
342
|
+
# Extract non-numeric characters at the beginning and end
|
|
343
|
+
match = re.match(r"(^\D*)(?:[\d,.]+)?(\D*$)", num_str)
|
|
344
|
+
if not match:
|
|
345
|
+
return None
|
|
346
|
+
|
|
347
|
+
prefix, suffix = match.groups()
|
|
348
|
+
|
|
349
|
+
# Check if the string is a valid number
|
|
350
|
+
if not __validate_numstr(num_str):
|
|
351
|
+
return None
|
|
352
|
+
else:
|
|
353
|
+
cleaned_numeric_part = __validate_numstr(num_str, True)
|
|
354
|
+
|
|
355
|
+
# Determine if the number contains a decimal
|
|
356
|
+
if '.' in cleaned_numeric_part:
|
|
357
|
+
integer_part, decimal_part = cleaned_numeric_part.split('.')
|
|
358
|
+
integer_part = int(integer_part) # Convert to int to remove leading zeros if any
|
|
359
|
+
else:
|
|
360
|
+
integer_part, decimal_part = int(cleaned_numeric_part), None
|
|
361
|
+
|
|
362
|
+
# Format the integer part with the specified separator
|
|
363
|
+
formatted_integer_part = f"{integer_part:,}".replace(",", sep)
|
|
364
|
+
|
|
365
|
+
# Reassemble the formatted number with the original non-numeric characters
|
|
366
|
+
if decimal_part is not None:
|
|
367
|
+
formatted_number = f"{formatted_integer_part}.{decimal_part}"
|
|
368
|
+
else:
|
|
369
|
+
formatted_number = formatted_integer_part
|
|
370
|
+
|
|
371
|
+
return f"{prefix}{formatted_number}{suffix}"
|
|
372
|
+
|
|
373
|
+
def formatDecimal(n: Union[int, float, str], place: int = 5) -> Optional[str]:
|
|
374
|
+
"""
|
|
375
|
+
Formats a given number to ensure it has a fixed number of decimal places.
|
|
376
|
+
|
|
377
|
+
Parameters:
|
|
378
|
+
──────────────────────────
|
|
379
|
+
n (Union[int, float, str]): The input number as a string, integer, or float.
|
|
380
|
+
place (int, optional): The number of decimal places to enforce. Default is **5**.
|
|
381
|
+
|
|
382
|
+
Returns:
|
|
383
|
+
──────────────────────────
|
|
384
|
+
Optional[str]: The formatted number with the specified decimal places.
|
|
385
|
+
Returns **None** if the input is invalid.
|
|
386
|
+
"""
|
|
387
|
+
# Ensure the input is a string
|
|
388
|
+
n = str(n)
|
|
389
|
+
n = " ".join(n.split())
|
|
390
|
+
|
|
391
|
+
# Check if the string is a valid number
|
|
392
|
+
if not __validate_numstr(n):
|
|
393
|
+
return None
|
|
394
|
+
|
|
395
|
+
place = int(place)
|
|
396
|
+
|
|
397
|
+
if '.' not in n:
|
|
398
|
+
n = f'{n}.{str(10**place).replace("1", "")}'
|
|
399
|
+
|
|
400
|
+
integer_part, decimal_part = n.split('.')
|
|
401
|
+
if len(decimal_part) < place:
|
|
402
|
+
decimal_part = decimal_part.ljust(place, '0')
|
|
403
|
+
elif len(decimal_part) > place:
|
|
404
|
+
decimal_part = decimal_part[:place]
|
|
405
|
+
|
|
406
|
+
return f"{integer_part}.{decimal_part}"
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
## Main Conversion Logic
|
|
412
|
+
##━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
# WORD-BASED NUMBER TO INTEGER: CONVERT CARDINAL NUMBERS IN WORD FORM (E.G., "TWENTY-FIVE") TO INTEGERS
|
|
416
|
+
#───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
|
|
417
|
+
def wordsToInt(s: str, thousands_sep: bool = False, sep: str = ","):
|
|
418
|
+
"""
|
|
419
|
+
Converts a spelled-out number in English into its integer equivalent.
|
|
420
|
+
|
|
421
|
+
This function processes cardinal numbers written in words and returns their
|
|
422
|
+
corresponding integer values. It supports numbers up to quintillions, including
|
|
423
|
+
multi-word formats and hyphenated numbers (e.g., "twenty-one"). Negative numbers
|
|
424
|
+
are also recognized if prefixed with "negative" or "minus".
|
|
425
|
+
|
|
426
|
+
The function can optionally format the output with thousands separators for better readability.
|
|
427
|
+
|
|
428
|
+
Parameters:
|
|
429
|
+
──────────────────────────
|
|
430
|
+
s (str): A string representing a number in English (e.g., "two hundred and fifty-six").
|
|
431
|
+
thousands_sep (bool, optional): If True, formats the output with a separator (default is False).
|
|
432
|
+
sep (str, optional): The character used as a thousands separator when `thousands_sep` is True (default is ',').
|
|
433
|
+
|
|
434
|
+
Returns:
|
|
435
|
+
──────────────────────────
|
|
436
|
+
int or None: The integer representation of the input if successfully parsed, otherwise None.
|
|
437
|
+
|
|
438
|
+
Notes:
|
|
439
|
+
──────────────────────────
|
|
440
|
+
- Does not process ordinal numbers (e.g., "first", "twenty-first").
|
|
441
|
+
- Assumes grammatically correct number formatting.
|
|
442
|
+
- Ignores the word "and" as it does not affect numerical value (e.g., "one hundred and five" = "one hundred five").
|
|
443
|
+
- Recognizes and processes compound words (e.g., "forty-two", "ninety-nine").
|
|
444
|
+
- If an unrecognized word is found, the function returns `None`.
|
|
445
|
+
"""
|
|
446
|
+
if not s:
|
|
447
|
+
return None
|
|
448
|
+
|
|
449
|
+
# Detect negative at the front
|
|
450
|
+
is_negative = False
|
|
451
|
+
s = s.strip().lower()
|
|
452
|
+
if s.startswith("negative "):
|
|
453
|
+
is_negative = True
|
|
454
|
+
s = s.replace("negative ", "", 1)
|
|
455
|
+
elif s.startswith("minus "):
|
|
456
|
+
is_negative = True
|
|
457
|
+
s = s.replace("minus ", "", 1)
|
|
458
|
+
|
|
459
|
+
# If it's an ordinal phrase, we skip it by returning None
|
|
460
|
+
token = __parseNumericToken(s)
|
|
461
|
+
if token and ordinalWordsToInt(token):
|
|
462
|
+
return None
|
|
463
|
+
|
|
464
|
+
units_dict = {word: i + 1 for i, word in enumerate(_UNIT_DIGITS_WORDS)}
|
|
465
|
+
tens_dict = {word: (i + 1) * 10 for i, word in enumerate(_TENS_MULTIPLES_WORDS)}
|
|
466
|
+
teens_dict = {word: i + 11 for i, word in enumerate(_TEEN_NUMERALS_WORDS)}
|
|
467
|
+
|
|
468
|
+
# Pre-compute some compound forms (e.g. "twenty-hundred" though not standard, etc.)
|
|
469
|
+
for mult in _MULTIPLIERS:
|
|
470
|
+
for word_list in [_UNIT_DIGITS_WORDS, _TENS_MULTIPLES_WORDS]:
|
|
471
|
+
for w in word_list:
|
|
472
|
+
compound = w + "-" + mult
|
|
473
|
+
if w in tens_dict:
|
|
474
|
+
units_dict[compound] = tens_dict[w] * _MULTIPLIERS[mult]
|
|
475
|
+
else:
|
|
476
|
+
units_dict[compound] = units_dict[w] * _MULTIPLIERS[mult]
|
|
477
|
+
|
|
478
|
+
# Split out hyphens/spaces, skip "and" tokens
|
|
479
|
+
words_list = re.split(r"[\s-]+", s)
|
|
480
|
+
words_list = [w for w in words_list if w != "and"] # skip "and"
|
|
481
|
+
|
|
482
|
+
number = 0
|
|
483
|
+
temp_number = 0
|
|
484
|
+
|
|
485
|
+
for word in words_list:
|
|
486
|
+
if word in units_dict:
|
|
487
|
+
temp_number += units_dict[word]
|
|
488
|
+
elif word in teens_dict:
|
|
489
|
+
temp_number += teens_dict[word]
|
|
490
|
+
elif word in tens_dict:
|
|
491
|
+
temp_number += tens_dict[word]
|
|
492
|
+
elif word in _MULTIPLIERS:
|
|
493
|
+
# For 'hundred', multiply the existing temp_number by 100
|
|
494
|
+
# For thousand/million/etc., multiply and then "commit" to number
|
|
495
|
+
temp_number *= _MULTIPLIERS[word]
|
|
496
|
+
if _MULTIPLIERS[word] >= 1000:
|
|
497
|
+
number += temp_number
|
|
498
|
+
temp_number = 0
|
|
499
|
+
else:
|
|
500
|
+
# If unknown word, fail
|
|
501
|
+
return None
|
|
502
|
+
|
|
503
|
+
number += temp_number
|
|
504
|
+
|
|
505
|
+
if number == 0:
|
|
506
|
+
return None
|
|
507
|
+
|
|
508
|
+
if is_negative:
|
|
509
|
+
number = -number
|
|
510
|
+
if thousands_sep:
|
|
511
|
+
number = insertSep(number, sep=sep)
|
|
512
|
+
return number
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
# WORD-BASED NUMBER TO INTEGER: CONVERT ORDINAL NUMBERS IN WORD FORM (E.G., "TWENTY-FIRST") TO INTEGERS
|
|
516
|
+
#───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
|
|
517
|
+
def ordinalWordsToInt(s: str, to_num: bool = False, thousands_sep: bool = False, sep: str = ","):
|
|
518
|
+
"""
|
|
519
|
+
Converts a spelled-out ordinal number in English to its numeric form.
|
|
520
|
+
|
|
521
|
+
This function processes ordinal words and converts them into either a numeric
|
|
522
|
+
string with a suffix (e.g., "1st", "21st") or an integer (if `to_num=True`).
|
|
523
|
+
It handles multi-word ordinals, hyphenated ordinals, and recognizes negative
|
|
524
|
+
ordinal numbers when prefixed with "negative" or "minus".
|
|
525
|
+
|
|
526
|
+
Parameters:
|
|
527
|
+
──────────────────────────
|
|
528
|
+
s (str): A string representing an ordinal number in English
|
|
529
|
+
(e.g., "twenty-first", "hundredth").
|
|
530
|
+
to_num (bool, optional): If True, returns the ordinal as an integer instead
|
|
531
|
+
of a string with a suffix. Default is False.
|
|
532
|
+
thousands_sep (bool, optional): If True, formats the output with a separator (default is False).
|
|
533
|
+
sep (str, optional): The character used as a thousands separator when `thousands_sep` is True (default is ',').
|
|
534
|
+
|
|
535
|
+
Returns:
|
|
536
|
+
──────────────────────────
|
|
537
|
+
str or int or None: The numeric ordinal representation as a string with a suffix
|
|
538
|
+
(e.g., "21st") or as an integer (if `to_num=True`), or
|
|
539
|
+
None if parsing fails.
|
|
540
|
+
"""
|
|
541
|
+
# Define lists of spelled-out cardinal and ordinal numbers and combine both lists
|
|
542
|
+
cardinal_numbers = ['zero', *_UNIT_DIGITS_WORDS, *_TEEN_NUMERALS_WORDS, *_TENS_MULTIPLES_WORDS]
|
|
543
|
+
ordinal_numbers = list(_ORDINAL_MAPPING.keys())
|
|
544
|
+
number_words = cardinal_numbers + ordinal_numbers
|
|
545
|
+
|
|
546
|
+
# Build a regex pattern that matches a hyphen only if it is between two valid number words
|
|
547
|
+
pattern = r'\b(' + '|'.join(number_words) + r')-(' + '|'.join(number_words) + r')\b'
|
|
548
|
+
|
|
549
|
+
# s = re.sub(r'(\w)-(\w)', r'\1 \2', s)
|
|
550
|
+
|
|
551
|
+
# Replaces hyphens only if they are between spelled-out cardinal or ordinal number words.
|
|
552
|
+
s = re.sub(pattern, r'\1 \2', s, flags=re.IGNORECASE)
|
|
553
|
+
|
|
554
|
+
if not s:
|
|
555
|
+
return None
|
|
556
|
+
|
|
557
|
+
# Detect negative at the front (rare for ordinals, but let's allow it)
|
|
558
|
+
is_negative = False
|
|
559
|
+
s = s.strip().lower()
|
|
560
|
+
if s.startswith("negative "):
|
|
561
|
+
is_negative = True
|
|
562
|
+
s = s.replace("negative ", "", 1)
|
|
563
|
+
elif s.startswith("minus "):
|
|
564
|
+
is_negative = True
|
|
565
|
+
s = s.replace("minus ", "", 1)
|
|
566
|
+
|
|
567
|
+
def _strip_num(ordinal_number):
|
|
568
|
+
"""
|
|
569
|
+
From '21st', '32nd', etc. -> integer
|
|
570
|
+
"""
|
|
571
|
+
ordinal_match = re.match(r'^-?(\d+)(st|nd|rd|th)$', ordinal_number)
|
|
572
|
+
return int(ordinal_match.group(1)) if ordinal_match else None
|
|
573
|
+
|
|
574
|
+
def _simple_ord(tok):
|
|
575
|
+
"""
|
|
576
|
+
Simple single-word ordinal: 'first' -> '1st', 'second' -> '2nd', etc.
|
|
577
|
+
"""
|
|
578
|
+
# Attempt direct mapping
|
|
579
|
+
if tok in _ORDINAL_MAPPING:
|
|
580
|
+
base_val = _ORDINAL_MAPPING[tok]
|
|
581
|
+
suffix = ordinalSuffix(base_val) # "st", "nd", "rd", "th"
|
|
582
|
+
# Return with or without numeric form
|
|
583
|
+
if to_num:
|
|
584
|
+
return -base_val if is_negative else base_val
|
|
585
|
+
else:
|
|
586
|
+
return f"{'-' if is_negative else ''}{base_val}{suffix}"
|
|
587
|
+
return None
|
|
588
|
+
|
|
589
|
+
def _complex_ord(tok):
|
|
590
|
+
"""
|
|
591
|
+
Handle multi-word ordinals: 'twenty first' -> '21st', 'one hundred and first' -> '101st'
|
|
592
|
+
"""
|
|
593
|
+
# We'll try removing the last word as an ordinal ending, then parse the front as cardinal
|
|
594
|
+
last_word_match = re.search(r'\b(\w+)\b$', tok)
|
|
595
|
+
if not last_word_match:
|
|
596
|
+
return None
|
|
597
|
+
last_word = last_word_match.group()
|
|
598
|
+
|
|
599
|
+
# Everything except the last word
|
|
600
|
+
front_string = re.sub(r'\s*\b\w+\b$', '', tok).strip()
|
|
601
|
+
|
|
602
|
+
# Convert front part to integer (cardinal)
|
|
603
|
+
front_number = wordsToInt(front_string)
|
|
604
|
+
|
|
605
|
+
if front_number is None:
|
|
606
|
+
return None
|
|
607
|
+
|
|
608
|
+
# Then interpret the last word as a single-word ordinal
|
|
609
|
+
last_ordinal_num = _simple_ord(last_word)
|
|
610
|
+
if last_ordinal_num is None:
|
|
611
|
+
return None
|
|
612
|
+
|
|
613
|
+
# If last_ordinal_num is integer (to_num=True) or a string with suffix
|
|
614
|
+
if isinstance(last_ordinal_num, int):
|
|
615
|
+
# If it is an integer, just sum
|
|
616
|
+
complete = front_number + last_ordinal_num
|
|
617
|
+
return complete if not is_negative else -complete
|
|
618
|
+
else:
|
|
619
|
+
# Otherwise it's something like "21st"
|
|
620
|
+
# Extract the integer portion from "21st"
|
|
621
|
+
stripped_val = _strip_num(last_ordinal_num)
|
|
622
|
+
if stripped_val is None:
|
|
623
|
+
return None
|
|
624
|
+
complete = front_number + stripped_val
|
|
625
|
+
final_suffix = ordinalSuffix(complete)
|
|
626
|
+
sign = "-" if is_negative else ""
|
|
627
|
+
return f"{sign}{complete}{final_suffix}"
|
|
628
|
+
|
|
629
|
+
# Parse to check if single or multiple words
|
|
630
|
+
# We'll use parseNumericToken to see if there's a direct token
|
|
631
|
+
# then handle single vs multi logic
|
|
632
|
+
token = __parseNumericToken(s)
|
|
633
|
+
if not token:
|
|
634
|
+
return None
|
|
635
|
+
|
|
636
|
+
# If more than one word, attempt complex
|
|
637
|
+
if len(token.split()) > 1:
|
|
638
|
+
check_number = _complex_ord(token)
|
|
639
|
+
if check_number is not None:
|
|
640
|
+
if to_num and thousands_sep:
|
|
641
|
+
return insertSep(check_number, sep=sep)
|
|
642
|
+
else:
|
|
643
|
+
return check_number
|
|
644
|
+
else:
|
|
645
|
+
# single word
|
|
646
|
+
check_number = _simple_ord(token)
|
|
647
|
+
if check_number is not None:
|
|
648
|
+
if to_num and thousands_sep:
|
|
649
|
+
return insertSep(check_number, sep=sep)
|
|
650
|
+
else:
|
|
651
|
+
return check_number
|
|
652
|
+
return None
|
|
653
|
+
|
|
654
|
+
# WORD-BASED NUMBER TO INTEGER: CONVERT NUMERIC STRINGS (E.G., "1,234", "42ND") TO INTEGERS
|
|
655
|
+
#───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
|
|
656
|
+
def stringToInt(s: str, to_str: bool = False, thousands_sep: bool = False, sep: str = ","):
|
|
657
|
+
"""
|
|
658
|
+
Converts a numeric string into an integer or string representation.
|
|
659
|
+
|
|
660
|
+
This function processes numeric strings containing digits, optionally formatted
|
|
661
|
+
with commas (e.g., "1,234"), and ordinal suffixes (e.g., "2nd"). It also handles
|
|
662
|
+
negative numbers prefixed with "negative" or "minus". The function returns an
|
|
663
|
+
integer by default but can return a string representation if `to_str=True`.
|
|
664
|
+
|
|
665
|
+
Parameters:
|
|
666
|
+
──────────────────────────
|
|
667
|
+
s (str): A numeric string, potentially containing commas or ordinal suffixes.
|
|
668
|
+
to_str (bool, optional): If True, returns the result as a string instead of an integer.
|
|
669
|
+
Default is False.
|
|
670
|
+
thousands_sep (bool, optional): If True, formats the output with a separator (default is False).
|
|
671
|
+
sep (str, optional): The character used as a thousands separator when `thousands_sep` is True (default is ',').
|
|
672
|
+
|
|
673
|
+
Returns:
|
|
674
|
+
──────────────────────────
|
|
675
|
+
int or str or None: The parsed integer or its string representation (if `to_str=True`).
|
|
676
|
+
Returns None if parsing fails.
|
|
677
|
+
"""
|
|
678
|
+
if not s:
|
|
679
|
+
return None
|
|
680
|
+
|
|
681
|
+
# Detect negative at the front
|
|
682
|
+
is_negative = False
|
|
683
|
+
number_str = s.strip()
|
|
684
|
+
if number_str.startswith("-"):
|
|
685
|
+
is_negative = True
|
|
686
|
+
number_str = number_str[1:].strip()
|
|
687
|
+
elif number_str.lower().startswith("negative "):
|
|
688
|
+
is_negative = True
|
|
689
|
+
number_str = number_str.lower().replace("negative ", "", 1)
|
|
690
|
+
elif number_str.lower().startswith("minus "):
|
|
691
|
+
is_negative = True
|
|
692
|
+
number_str = number_str.lower().replace("minus ", "", 1)
|
|
693
|
+
|
|
694
|
+
tokens = __parseNumericToken(number_str)
|
|
695
|
+
# if tokens and re.match(r'^-?\d+$', str(tokens)):
|
|
696
|
+
# val = int(tokens)
|
|
697
|
+
# val = -val if is_negative else val
|
|
698
|
+
# return str(val) if to_str else val
|
|
699
|
+
if tokens and re.match(r'^-?\d+$', str(tokens)):
|
|
700
|
+
val = int(tokens)
|
|
701
|
+
val = -val if is_negative else val
|
|
702
|
+
if thousands_sep and to_str:
|
|
703
|
+
return insertSep(val, sep=sep)
|
|
704
|
+
return str(val) if to_str else val
|
|
705
|
+
return None
|
|
706
|
+
|
|
707
|
+
|
|
708
|
+
|
|
709
|
+
|
|
710
|
+
# INTEGER TO WORD-BASED NUMBER: CONVERT INTEGER VALUES (E.G., 256) TO CARDINAL NUMBERS IN WORD FORM (E.G., "TWO HUNDRED FIFTY-SIX")
|
|
711
|
+
#───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
|
|
712
|
+
def intToWords(n: Union[int, float], thousands_sep: bool = False):
|
|
713
|
+
"""
|
|
714
|
+
Converts an integer or float into its spelled-out English words representation.
|
|
715
|
+
|
|
716
|
+
This function transforms numerical values into their corresponding English
|
|
717
|
+
words. It supports both positive and negative numbers, including large values
|
|
718
|
+
up to quintillions. Additionally, it can handle floating-point numbers by
|
|
719
|
+
spelling out both the integer and decimal portions separately.
|
|
720
|
+
|
|
721
|
+
If `thousands_sep=True`, large segments (e.g., thousands, millions, billions)
|
|
722
|
+
are separated by commas in the output.
|
|
723
|
+
|
|
724
|
+
Parameters:
|
|
725
|
+
──────────────────────────
|
|
726
|
+
n (int or float): The number to be converted to words.
|
|
727
|
+
thousands_sep (bool, optional): If True, inserts commas between large segments
|
|
728
|
+
for better readability. Default is False.
|
|
729
|
+
|
|
730
|
+
Returns:
|
|
731
|
+
──────────────────────────
|
|
732
|
+
str or None: The spelled-out English representation of the number, or None
|
|
733
|
+
if input is invalid.
|
|
734
|
+
"""
|
|
735
|
+
def _from_int(x):
|
|
736
|
+
if x == 0:
|
|
737
|
+
return "zero"
|
|
738
|
+
|
|
739
|
+
def one(num):
|
|
740
|
+
switcher = {
|
|
741
|
+
1: 'one', 2: 'two', 3: 'three', 4: 'four', 5: 'five',
|
|
742
|
+
6: 'six', 7: 'seven', 8: 'eight', 9: 'nine'
|
|
743
|
+
}
|
|
744
|
+
return switcher.get(num, '')
|
|
745
|
+
|
|
746
|
+
def two_less_20(num):
|
|
747
|
+
switcher = {
|
|
748
|
+
10: 'ten', 11: 'eleven', 12: 'twelve', 13: 'thirteen', 14: 'fourteen',
|
|
749
|
+
15: 'fifteen', 16: 'sixteen', 17: 'seventeen', 18: 'eighteen', 19: 'nineteen'
|
|
750
|
+
}
|
|
751
|
+
return switcher.get(num, '')
|
|
752
|
+
|
|
753
|
+
def ten(num):
|
|
754
|
+
switcher = {
|
|
755
|
+
2: 'twenty', 3: 'thirty', 4: 'forty', 5: 'fifty',
|
|
756
|
+
6: 'sixty', 7: 'seventy', 8: 'eighty', 9: 'ninety'
|
|
757
|
+
}
|
|
758
|
+
return switcher.get(num, '')
|
|
759
|
+
|
|
760
|
+
def two(num):
|
|
761
|
+
if not num:
|
|
762
|
+
return ''
|
|
763
|
+
elif num < 10:
|
|
764
|
+
return one(num)
|
|
765
|
+
elif num < 20:
|
|
766
|
+
return two_less_20(num)
|
|
767
|
+
else:
|
|
768
|
+
tenner = num // 10
|
|
769
|
+
rest = num % 10
|
|
770
|
+
return ten(tenner) + ('-' + one(rest) if rest else '')
|
|
771
|
+
|
|
772
|
+
def three(num):
|
|
773
|
+
hundred = num // 100
|
|
774
|
+
rest = num % 100
|
|
775
|
+
if hundred and rest:
|
|
776
|
+
return one(hundred) + ' hundred ' + two(rest)
|
|
777
|
+
elif hundred and not rest:
|
|
778
|
+
return one(hundred) + ' hundred'
|
|
779
|
+
else:
|
|
780
|
+
return two(rest)
|
|
781
|
+
|
|
782
|
+
# Break the number into billions, millions, thousands, and the remainder
|
|
783
|
+
# Extended to quintillions
|
|
784
|
+
# We'll do repeated modulus and division:
|
|
785
|
+
# e.g. 1,234,567,890,123 -> segments for trillions, billions, millions, thousands, rest
|
|
786
|
+
# We'll store all segments in ascending order, then build from largest to smallest for readability.
|
|
787
|
+
# For now, let's just go up to quintillions.
|
|
788
|
+
abs_num = abs(x)
|
|
789
|
+
|
|
790
|
+
quintillion = abs_num // 1000000000000000000
|
|
791
|
+
remainder_q = abs_num % 1000000000000000000
|
|
792
|
+
|
|
793
|
+
quadrillion = remainder_q // 1000000000000000
|
|
794
|
+
remainder_quad = remainder_q % 1000000000000000
|
|
795
|
+
|
|
796
|
+
trillion = remainder_quad // 1000000000000
|
|
797
|
+
remainder_tril = remainder_quad % 1000000000000
|
|
798
|
+
|
|
799
|
+
billion = remainder_tril // 1000000000
|
|
800
|
+
remainder_bill = remainder_tril % 1000000000
|
|
801
|
+
|
|
802
|
+
million = remainder_bill // 1000000
|
|
803
|
+
remainder_mill = remainder_bill % 1000000
|
|
804
|
+
|
|
805
|
+
thousand = remainder_mill // 1000
|
|
806
|
+
remainder = remainder_mill % 1000
|
|
807
|
+
|
|
808
|
+
segments = []
|
|
809
|
+
if quintillion:
|
|
810
|
+
segments.append(three(quintillion) + " quintillion")
|
|
811
|
+
if quadrillion:
|
|
812
|
+
segments.append(three(quadrillion) + " quadrillion")
|
|
813
|
+
if trillion:
|
|
814
|
+
segments.append(three(trillion) + " trillion")
|
|
815
|
+
if billion:
|
|
816
|
+
segments.append(three(billion) + " billion")
|
|
817
|
+
if million:
|
|
818
|
+
segments.append(three(million) + " million")
|
|
819
|
+
if thousand:
|
|
820
|
+
segments.append(three(thousand) + " thousand")
|
|
821
|
+
if remainder:
|
|
822
|
+
segments.append(three(remainder))
|
|
823
|
+
|
|
824
|
+
result = ""
|
|
825
|
+
if segments:
|
|
826
|
+
if thousands_sep and len(segments) > 1:
|
|
827
|
+
result = ", ".join(seg for seg in segments if seg).strip()
|
|
828
|
+
else:
|
|
829
|
+
result = " ".join(seg for seg in segments if seg).strip()
|
|
830
|
+
else:
|
|
831
|
+
result = "zero"
|
|
832
|
+
|
|
833
|
+
# Attach negative sign if needed
|
|
834
|
+
if x < 0:
|
|
835
|
+
result = "negative " + result
|
|
836
|
+
|
|
837
|
+
return result
|
|
838
|
+
|
|
839
|
+
def _from_float(num):
|
|
840
|
+
"""
|
|
841
|
+
Very basic approach to handle floats by splitting at the decimal.
|
|
842
|
+
"""
|
|
843
|
+
whole_str, decimal_str = str(num).split(".")
|
|
844
|
+
whole_part = _from_int(int(whole_str))
|
|
845
|
+
# Convert each digit in the decimal part to words, or parse the entire decimal as an integer:
|
|
846
|
+
# "45" -> "forty-five" or "four five"
|
|
847
|
+
dec_int = int(decimal_str)
|
|
848
|
+
decimal_words = _from_int(dec_int)
|
|
849
|
+
|
|
850
|
+
return f"{whole_part} point {decimal_words}"
|
|
851
|
+
|
|
852
|
+
n_str = str(n)
|
|
853
|
+
if "." in n_str:
|
|
854
|
+
try:
|
|
855
|
+
float_val = float(n_str)
|
|
856
|
+
return _from_float(float_val)
|
|
857
|
+
except ValueError:
|
|
858
|
+
return None
|
|
859
|
+
else:
|
|
860
|
+
try:
|
|
861
|
+
int_val = int(n_str)
|
|
862
|
+
return _from_int(int_val)
|
|
863
|
+
except ValueError:
|
|
864
|
+
return None
|
|
865
|
+
|
|
866
|
+
|
|
867
|
+
# INTEGER TO WORD-BASED NUMBER: CONVERT INTEGER VALUES (E.G., 21) TO ORDINAL NUMBERS IN WORD FORM (E.G., "TWENTY-FIRST")
|
|
868
|
+
#───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
|
|
869
|
+
def intToOrdinalWords(n: int):
|
|
870
|
+
"""
|
|
871
|
+
Converts an integer into its ordinal spelled-out English form.
|
|
872
|
+
|
|
873
|
+
This function takes an integer and returns its ordinal representation in words.
|
|
874
|
+
It correctly handles standard English transformations for ordinal numbers,
|
|
875
|
+
including irregular forms like "first", "second", "third", "twelfth", and
|
|
876
|
+
suffix changes for numbers ending in "-y" (e.g., "twenty" -> "twentieth").
|
|
877
|
+
|
|
878
|
+
The function supports both positive and negative numbers.
|
|
879
|
+
|
|
880
|
+
Parameters:
|
|
881
|
+
──────────────────────────
|
|
882
|
+
n (int): The integer to be converted into ordinal words.
|
|
883
|
+
|
|
884
|
+
Returns:
|
|
885
|
+
──────────────────────────
|
|
886
|
+
str or None: The ordinal representation of the number as a string,
|
|
887
|
+
or None if conversion fails.
|
|
888
|
+
"""
|
|
889
|
+
words = intToWords(n, thousands_sep=False)
|
|
890
|
+
if not words:
|
|
891
|
+
return None
|
|
892
|
+
|
|
893
|
+
# We'll split the spelled-out form, then transform the last word.
|
|
894
|
+
# Note: This is a simplistic approach and might need special handling for multi-segment final words.
|
|
895
|
+
word_parts = words.split()
|
|
896
|
+
if not word_parts:
|
|
897
|
+
return None
|
|
898
|
+
|
|
899
|
+
# Identify the negative sign if it exists
|
|
900
|
+
is_negative = False
|
|
901
|
+
if word_parts[0] == "negative":
|
|
902
|
+
is_negative = True
|
|
903
|
+
word_parts = word_parts[1:] # remove "negative"
|
|
904
|
+
|
|
905
|
+
last_word = word_parts[-1]
|
|
906
|
+
|
|
907
|
+
def _replace_end(full_word, old_end, new_end):
|
|
908
|
+
return full_word[: -len(old_end)] + new_end if full_word.endswith(old_end) else full_word
|
|
909
|
+
|
|
910
|
+
# We'll do a set of special transformations:
|
|
911
|
+
# one -> first, two -> second, three -> third, etc.
|
|
912
|
+
# This is a subset of patterns.
|
|
913
|
+
if last_word.endswith("one"):
|
|
914
|
+
word_parts[-1] = _replace_end(last_word, "one", "first")
|
|
915
|
+
elif last_word.endswith("two"):
|
|
916
|
+
word_parts[-1] = _replace_end(last_word, "two", "second")
|
|
917
|
+
elif last_word.endswith("three"):
|
|
918
|
+
word_parts[-1] = _replace_end(last_word, "three", "third")
|
|
919
|
+
elif last_word.endswith("five"):
|
|
920
|
+
word_parts[-1] = _replace_end(last_word, "five", "fifth")
|
|
921
|
+
elif last_word.endswith("eight"):
|
|
922
|
+
word_parts[-1] = _replace_end(last_word, "eight", "eighth")
|
|
923
|
+
elif last_word.endswith("nine"):
|
|
924
|
+
word_parts[-1] = _replace_end(last_word, "nine", "ninth")
|
|
925
|
+
elif last_word.endswith("twelve"):
|
|
926
|
+
word_parts[-1] = _replace_end(last_word, "twelve", "twelfth")
|
|
927
|
+
elif last_word.endswith("y"):
|
|
928
|
+
word_parts[-1] = _replace_end(last_word, "y", "ieth") # e.g. "twenty" -> "twentieth", "thirty" -> "thirtieth"
|
|
929
|
+
elif last_word.endswith("teen"):
|
|
930
|
+
word_parts[-1] = _replace_end(last_word, "teen", "teenth") # e.g. "fourteen" -> "fourteenth"
|
|
931
|
+
else:
|
|
932
|
+
word_parts[-1] = word_parts[-1] + "th" # Generic
|
|
933
|
+
|
|
934
|
+
if is_negative:
|
|
935
|
+
return "negative " + " ".join(word_parts)
|
|
936
|
+
else:
|
|
937
|
+
return " ".join(word_parts)
|
|
938
|
+
|
|
939
|
+
|
|
940
|
+
|
|
941
|
+
|
|
942
|
+
# ORDINAL NUMBER UTILITIES: EXTRACT THE APPROPRIATE SUFFIX ("ST", "ND", "RD", "TH") FOR AN INTEGER
|
|
943
|
+
#───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
|
|
944
|
+
def ordinalSuffix(n: int):
|
|
945
|
+
"""
|
|
946
|
+
Determines the appropriate English ordinal suffix for a given integer.
|
|
947
|
+
|
|
948
|
+
This function returns the correct ordinal suffix ("st", "nd", "rd", "th")
|
|
949
|
+
based on standard English rules. It properly accounts for special cases
|
|
950
|
+
where numbers ending in 11, 12, or 13 always take "th".
|
|
951
|
+
|
|
952
|
+
Parameters:
|
|
953
|
+
──────────────────────────
|
|
954
|
+
n (int): The integer for which to determine the ordinal suffix.
|
|
955
|
+
|
|
956
|
+
Returns:
|
|
957
|
+
──────────────────────────
|
|
958
|
+
str: The appropriate ordinal suffix ('st', 'nd', 'rd', or 'th').
|
|
959
|
+
"""
|
|
960
|
+
last_two = abs(n) % 100
|
|
961
|
+
last_digit = abs(n) % 10
|
|
962
|
+
if last_two in (11, 12, 13):
|
|
963
|
+
return "th"
|
|
964
|
+
else:
|
|
965
|
+
if last_digit == 1:
|
|
966
|
+
return "st"
|
|
967
|
+
elif last_digit == 2:
|
|
968
|
+
return "nd"
|
|
969
|
+
elif last_digit == 3:
|
|
970
|
+
return "rd"
|
|
971
|
+
else:
|
|
972
|
+
return "th"
|
|
973
|
+
|
|
974
|
+
# ORDINAL NUMBER UTILITIES: REMOVE THE ORDINAL ENDING FROM A WORD-BASED NUMBER (E.G., "TWENTIETH" → "TWENTY")
|
|
975
|
+
#───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
|
|
976
|
+
def stripOrdinalSuffix(s: str):
|
|
977
|
+
"""
|
|
978
|
+
Strips the ordinal suffix from a spelled-out ordinal number, returning the base
|
|
979
|
+
cardinal form and the suffix separately.
|
|
980
|
+
|
|
981
|
+
This function identifies and removes ordinal suffixes from spelled-out ordinal
|
|
982
|
+
numbers (e.g., "twentieth" → "twenty", "twenty-first" → "twenty-one"). It returns
|
|
983
|
+
a tuple containing the base cardinal number as a string and the ordinal suffix.
|
|
984
|
+
|
|
985
|
+
Parameters:
|
|
986
|
+
──────────────────────────
|
|
987
|
+
s (str): A spelled-out ordinal number (e.g., "seventh", "thirty-second").
|
|
988
|
+
|
|
989
|
+
Returns:
|
|
990
|
+
──────────────────────────
|
|
991
|
+
tuple or None: A tuple containing the base cardinal number (str) and its
|
|
992
|
+
ordinal suffix (str), or None if no ordinal suffix is detected.
|
|
993
|
+
"""
|
|
994
|
+
number_str = s
|
|
995
|
+
suffix = None
|
|
996
|
+
for pattern, (replacement, suffix_to_remove) in _WORD_BASED_PATTERNS_RE.items():
|
|
997
|
+
if re.search(pattern, s, flags=re.IGNORECASE):
|
|
998
|
+
suffix = suffix_to_remove
|
|
999
|
+
number_str = re.sub(pattern, replacement, number_str, flags=re.IGNORECASE)
|
|
1000
|
+
break # Stop at first match
|
|
1001
|
+
if suffix is None:
|
|
1002
|
+
return None
|
|
1003
|
+
return (number_str, suffix)
|
|
1004
|
+
|
|
1005
|
+
|
|
1006
|
+
|
|
1007
|
+
|
|
1008
|
+
# GENERAL NUMBER EXTRACTION: IDENTIFY AND CONVERT ALL NUMERIC VALUES (CARDINAL OR ORDINAL) FROM A STRING
|
|
1009
|
+
#───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
|
|
1010
|
+
def extractNumericValue(s: str, allnum: bool = True):
|
|
1011
|
+
"""
|
|
1012
|
+
Extracts and converts all numeric values (cardinal or ordinal) from a string into a list of integers.
|
|
1013
|
+
|
|
1014
|
+
This function searches for multiple numeric values, whether they are:
|
|
1015
|
+
- Digit-based (e.g., "42", "1,234")
|
|
1016
|
+
- Ordinals (e.g., "third", "42nd")
|
|
1017
|
+
- Spelled-out numbers (e.g., "twenty-five")
|
|
1018
|
+
|
|
1019
|
+
If a negative indicator ("negative" or "minus") is present at the start of the string,
|
|
1020
|
+
it applies negativity only to the **first** detected number.
|
|
1021
|
+
|
|
1022
|
+
Parameters:
|
|
1023
|
+
──────────────────────────
|
|
1024
|
+
s (str): A string potentially containing multiple numeric values.
|
|
1025
|
+
allnum (bool, optional): Determines whether to return **all** numeric matches or just the **first** one.
|
|
1026
|
+
- `True` (default): Returns a list of all numbers found in the string.
|
|
1027
|
+
- `False`: Returns only the first number found.
|
|
1028
|
+
|
|
1029
|
+
Returns:
|
|
1030
|
+
──────────────────────────
|
|
1031
|
+
list[int] or None:
|
|
1032
|
+
- A list of parsed integers if numeric values are found.
|
|
1033
|
+
- None if no numbers are detected.
|
|
1034
|
+
"""
|
|
1035
|
+
if not s:
|
|
1036
|
+
return None
|
|
1037
|
+
|
|
1038
|
+
# Handle negative at the start of the string
|
|
1039
|
+
is_negative = False
|
|
1040
|
+
string = s.strip().lower()
|
|
1041
|
+
if string.startswith("negative "):
|
|
1042
|
+
is_negative = True
|
|
1043
|
+
string = string.replace("negative ", "", 1)
|
|
1044
|
+
elif string.startswith("minus "):
|
|
1045
|
+
is_negative = True
|
|
1046
|
+
string = string.replace("minus ", "", 1)
|
|
1047
|
+
|
|
1048
|
+
# tokens = __parseNumericToken(s, first_only=False) # Get all matches
|
|
1049
|
+
tokens = __parseNumericToken(s, first_only=__switch(allnum), wrap_single=True) # Get all matches if allnum == True. The switch function swithes allnum boolen to False.
|
|
1050
|
+
if not tokens:
|
|
1051
|
+
return None
|
|
1052
|
+
|
|
1053
|
+
# if not isinstance(tokens, list):
|
|
1054
|
+
# tokens=[tokens]
|
|
1055
|
+
|
|
1056
|
+
def _check_and_return(num):
|
|
1057
|
+
return num if isinstance(num, int) else None
|
|
1058
|
+
|
|
1059
|
+
# Define functions to apply for conversion
|
|
1060
|
+
funcs_and_kwargs = [
|
|
1061
|
+
(ordinalWordsToInt, {'to_num': True}),
|
|
1062
|
+
(stringToInt, {'to_str': False}),
|
|
1063
|
+
(wordsToInt, {}),
|
|
1064
|
+
]
|
|
1065
|
+
|
|
1066
|
+
# Process all tokens
|
|
1067
|
+
parsed_numbers = []
|
|
1068
|
+
for token in tokens:
|
|
1069
|
+
for func, kwargs in funcs_and_kwargs:
|
|
1070
|
+
result = func(token, **kwargs)
|
|
1071
|
+
number = _check_and_return(result)
|
|
1072
|
+
if number is not None:
|
|
1073
|
+
# Apply negativity only to the first detected number
|
|
1074
|
+
if is_negative and not parsed_numbers: # Only negate the **first** number found
|
|
1075
|
+
number = -abs(number)
|
|
1076
|
+
parsed_numbers.append(number)
|
|
1077
|
+
break # Stop trying other functions once parsed successfully
|
|
1078
|
+
|
|
1079
|
+
# Return single integer if only one number is found, otherwise return list
|
|
1080
|
+
if not parsed_numbers:
|
|
1081
|
+
return None
|
|
1082
|
+
return parsed_numbers[0] if len(parsed_numbers) == 1 else parsed_numbers
|
|
1083
|
+
|
|
1084
|
+
|
|
1085
|
+
|
|
1086
|
+
# ROMAN NUMERALS: CONVERT ROMAN NUMERALS (E.G., "XIV") TO INTEGERS
|
|
1087
|
+
#───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
|
|
1088
|
+
def romanToInt(s: str, to_str: bool = False):
|
|
1089
|
+
"""
|
|
1090
|
+
Converts a Roman numeral string into its integer value.
|
|
1091
|
+
|
|
1092
|
+
This function parses and converts valid Roman numeral strings into their
|
|
1093
|
+
corresponding integer values while ensuring strict adherence to Roman numeral
|
|
1094
|
+
rules. It handles both standard and subtractive notation (e.g., 'XIV' = 14,
|
|
1095
|
+
'MCMXCIV' = 1994) and validates input to prevent incorrect sequences.
|
|
1096
|
+
|
|
1097
|
+
Parameters:
|
|
1098
|
+
──────────────────────────
|
|
1099
|
+
- s (*str*):
|
|
1100
|
+
- A valid Roman numeral string (e.g., 'XIV', 'MCMXCIV').
|
|
1101
|
+
|
|
1102
|
+
- to_str (*bool, optional*):
|
|
1103
|
+
- If `True`, returns the result as a string instead of an integer.
|
|
1104
|
+
- Default is `False`.
|
|
1105
|
+
|
|
1106
|
+
Returns:
|
|
1107
|
+
──────────────────────────
|
|
1108
|
+
- (*int | str | None*):
|
|
1109
|
+
- The integer representation of the Roman numeral.
|
|
1110
|
+
- If `to_str=True`, returns the value as a string.
|
|
1111
|
+
- Returns `None` if the input is invalid.
|
|
1112
|
+
"""
|
|
1113
|
+
num = __parseRomanNumeral(s)
|
|
1114
|
+
return str(num) if num is not None and to_str else num
|
|
1115
|
+
|
|
1116
|
+
|
|
1117
|
+
# ROMAN NUMERALS: CONVERT ROMAN NUMERALS (E.G., "XIV") TO CARDINAL NUMBERS IN WORD FORM (E.G., "FOURTEEN")
|
|
1118
|
+
#───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
|
|
1119
|
+
def romanToWords(s: str):
|
|
1120
|
+
"""
|
|
1121
|
+
Converts a Roman numeral string into its spelled-out English words representation.
|
|
1122
|
+
|
|
1123
|
+
This function first converts a Roman numeral into an integer and then transforms
|
|
1124
|
+
it into its full English word equivalent. It ensures strict adherence to Roman
|
|
1125
|
+
numeral rules and returns a readable word-based representation.
|
|
1126
|
+
|
|
1127
|
+
Parameters:
|
|
1128
|
+
──────────────────────────
|
|
1129
|
+
- s (*str*):
|
|
1130
|
+
- A valid Roman numeral string (e.g., 'XIV', 'MCMXCIV').
|
|
1131
|
+
|
|
1132
|
+
Returns:
|
|
1133
|
+
──────────────────────────
|
|
1134
|
+
- (*str | None*):
|
|
1135
|
+
- The spelled-out English words for the given Roman numeral.
|
|
1136
|
+
- Returns `None` if the input is invalid.
|
|
1137
|
+
"""
|
|
1138
|
+
num = __parseRomanNumeral(s)
|
|
1139
|
+
return intToWords(num) if num is not None else None
|
|
1140
|
+
|
|
1141
|
+
|
|
1142
|
+
|
|
1143
|
+
|
|
1144
|
+
__all__ = [
|
|
1145
|
+
# "replaceNumericValue",
|
|
1146
|
+
"wordsToInt",
|
|
1147
|
+
"ordinalSuffix",
|
|
1148
|
+
"intToWords",
|
|
1149
|
+
"intToOrdinalWords",
|
|
1150
|
+
"stripOrdinalSuffix",
|
|
1151
|
+
"ordinalWordsToInt",
|
|
1152
|
+
"stringToInt",
|
|
1153
|
+
"extractNumericValue",
|
|
1154
|
+
"romanToWords",
|
|
1155
|
+
"romanToInt",
|
|
1156
|
+
"insertSep",
|
|
1157
|
+
"formatDecimal",
|
|
1158
|
+
]
|
|
1159
|
+
|
|
1160
|
+
|
|
1161
|
+
|
|
1162
|
+
|
|
1163
|
+
|
|
1164
|
+
|
|
1165
|
+
|
|
1166
|
+
|
|
1167
|
+
|
|
1168
|
+
|
|
1169
|
+
|
|
1170
|
+
|
|
1171
|
+
|
|
1172
|
+
|
|
1173
|
+
|
|
1174
|
+
|
|
1175
|
+
|
|
1176
|
+
|
|
1177
|
+
|
|
1178
|
+
|
|
1179
|
+
|
|
1180
|
+
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: numbr
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: A comprehensive Python library for parsing and converting numbers between numeric, word, and ordinal formats.
|
|
5
|
+
Home-page: https://github.com/cedricmoorejr/numbr/tree/v1.0.0
|
|
6
|
+
Author: Cedric Moore Jr.
|
|
7
|
+
Author-email: cedricmoorejunior5@gmail.com
|
|
8
|
+
License: MIT
|
|
9
|
+
Classifier: Programming Language :: Python :: 3
|
|
10
|
+
Classifier: Operating System :: Microsoft :: Windows
|
|
11
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
12
|
+
Classifier: Operating System :: MacOS :: MacOS X
|
|
13
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Natural Language :: English
|
|
16
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
17
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
21
|
+
Requires-Python: >=3.6
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
|
|
24
|
+
# numbr
|
|
25
|
+
|
|
26
|
+
**numbr** is a Python library designed for parsing and converting numbers written in English. It simplifies working with numbers by converting between spelled-out forms, ordinal forms, and numeric representations.
|
|
27
|
+
|
|
28
|
+
---
|
|
29
|
+
|
|
30
|
+
## Key Features
|
|
31
|
+
|
|
32
|
+
- Convert spelled-out cardinal numbers into integers (e.g., `"one hundred twenty-three"` → `123`).
|
|
33
|
+
- Convert integers into their spelled-out English words (e.g., `123` → `"one hundred twenty-three"`).
|
|
34
|
+
- Convert ordinal words to numeric ordinals (e.g., `"twenty-first"` → `"21st"` or `21`).
|
|
35
|
+
- Extract numeric values from text strings.
|
|
36
|
+
- Handle negative numbers, hyphenated numbers, and large numbers (up to quintillions).
|
|
37
|
+
|
|
38
|
+
---
|
|
39
|
+
|
|
40
|
+
## Installation
|
|
41
|
+
|
|
42
|
+
Install `numbr` using `pip`:
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
pip install numbr
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
---
|
|
49
|
+
|
|
50
|
+
## Usage Examples
|
|
51
|
+
|
|
52
|
+
### Convert Words to Integer
|
|
53
|
+
|
|
54
|
+
```python
|
|
55
|
+
import numbr
|
|
56
|
+
|
|
57
|
+
print(numbr.wordsToInt("one thousand two hundred thirty-four"))
|
|
58
|
+
# Output: 1234
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
### Convert Integer to Words
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
print(numbr.intToWords(5678))
|
|
65
|
+
# Output: "five thousand six hundred seventy-eight"
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
### Convert Ordinal Words to Numeric Form
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
print(numbr.ordinalWordsToInt("forty-second"))
|
|
72
|
+
# Output: "42nd"
|
|
73
|
+
|
|
74
|
+
print(numbr.ordinalWordsToInt("forty-second", to_num=True))
|
|
75
|
+
# Output: 42
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
### Extract Numeric Values from Strings
|
|
79
|
+
|
|
80
|
+
```python
|
|
81
|
+
print(numbr.extractNumericValue("I have twenty apples and 13 oranges."))
|
|
82
|
+
# Output: 20
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
---
|
|
86
|
+
|
|
87
|
+
## Terminology
|
|
88
|
+
|
|
89
|
+
- **Cardinal Numbers**: Represent quantity (e.g., "one", "twenty-five", "1,234").
|
|
90
|
+
- **Ordinal Numbers**: Represent position or order (e.g., "first", "twenty-first", "3rd").
|
|
91
|
+
- **Ordinal Suffix**: Letters added to numbers indicating position ("st", "nd", "rd", "th").
|
|
92
|
+
|
|
93
|
+
---
|
|
94
|
+
|
|
95
|
+
## License
|
|
96
|
+
|
|
97
|
+
This project is licensed under the MIT License. See the [LICENSE](LICENSE) file for details.
|
|
98
|
+
|
|
99
|
+
---
|
|
100
|
+
|
|
101
|
+
## Contributing
|
|
102
|
+
|
|
103
|
+
Contributions are welcome! Feel free to open issues or submit pull requests to improve `numbr`.
|
|
104
|
+
|
|
105
|
+
---
|
|
106
|
+
|
|
107
|
+
## Contact
|
|
108
|
+
|
|
109
|
+
For questions or feedback, please open an issue on the project's GitHub repository.
|
|
110
|
+
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
numbr/__init__.py,sha256=ih6LAoWE2-ZfrkAWPcUIyYTh2VMVF9uysdHZSh7FVeQ,991
|
|
2
|
+
numbr/engine.py,sha256=pnvuEGaihKJe-5-2O3iuwbSMVGgoOvBPDom1fcvofJ0,53157
|
|
3
|
+
numbr-1.0.0.dist-info/METADATA,sha256=2CwNXybHv-LaK6ELNybV3gprG-QhdW1znknmyTn46XA,3101
|
|
4
|
+
numbr-1.0.0.dist-info/WHEEL,sha256=pkctZYzUS4AYVn6dJ-7367OJZivF2e8RA9b_ZBjif18,92
|
|
5
|
+
numbr-1.0.0.dist-info/top_level.txt,sha256=FoAqdARWQS0e47iqVRzh8HclyF_xyc_0saSAnjbfSco,6
|
|
6
|
+
numbr-1.0.0.dist-info/RECORD,,
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
numbr
|