numbr 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
numbr/__init__.py ADDED
@@ -0,0 +1,37 @@
1
+ # -*- coding: utf-8 -*-
2
+
3
+
4
+ from . import engine as __engine
5
+
6
+ __all__ = [
7
+ # "replaceNumericValue",
8
+ "wordsToInt",
9
+ "ordinalSuffix",
10
+ "intToWords",
11
+ "intToOrdinalWords",
12
+ "stripOrdinalSuffix",
13
+ "ordinalWordsToInt",
14
+ "stringToInt",
15
+ "extractNumericValue",
16
+ "romanToWords",
17
+ "romanToInt",
18
+ "formatDecimal",
19
+ "insertSep",
20
+ ]
21
+
22
+ # Reference functions using __engine alias
23
+ # replaceNumericValue = __engine.replaceNumericValue
24
+ wordsToInt = __engine.wordsToInt
25
+ ordinalSuffix = __engine.ordinalSuffix
26
+ intToWords = __engine.intToWords
27
+ intToOrdinalWords = __engine.intToOrdinalWords
28
+ stripOrdinalSuffix = __engine.stripOrdinalSuffix
29
+ ordinalWordsToInt = __engine.ordinalWordsToInt
30
+ stringToInt = __engine.stringToInt
31
+ extractNumericValue = __engine.extractNumericValue
32
+ romanToWords = __engine.romanToWords
33
+ romanToInt = __engine.romanToInt
34
+ formatDecimal = __engine.formatDecimal
35
+ insertSep = __engine.insertSep
36
+
37
+ del engine
numbr/engine.py ADDED
@@ -0,0 +1,1180 @@
1
+ # -*- coding: utf-8 -*-
2
+
3
+ #
4
+ # Understanding Number Terminology
5
+ # ─────────────────────────────────
6
+ # ┍━━━━━━━━━┯━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┯━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┑
7
+ # │ Example │ Type │ Description │
8
+ # ┝━━━━━━━━━┿━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┿━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┥
9
+ # │ four │ Cardinal number (word form) │ This is the written-out version of the number 4. │
10
+ # │ 4 │ Cardinal numeral (digit form) │ This is the numeric symbol representing "four." │
11
+ # │ fourth │ Ordinal number (word form) │ This is the written-out version of "4th," used to describe position. │
12
+ # │ 4th │ Ordinal numeral (digit form) │ This is the numeric way of writing an ordinal number. │
13
+ # │ IV │ Roman numeral │ This represents the number "4" in the Roman numeral system. │
14
+ # ┕━━━━━━━━━┷━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┷━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┙
15
+ #
16
+ # This module is designed to help developers work with numbers in natural language processing (NLP), data extraction,
17
+ # and automated text conversion.
18
+
19
+
20
+
21
+ import re
22
+ import inspect
23
+ from typing import Literal, Union, Optional
24
+ from functools import reduce
25
+
26
+
27
+
28
+ ## Module-Level Constants & Dictionaries
29
+ ##━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
30
+
31
+ # Matches spelled-out numbers, including cardinal and ordinal forms.
32
+ _NUM_ORDINAL_WRDS_RE = (
33
+ r'zero|one|two|three|four|five|six|seven|eight|nine|ten|'
34
+ r'eleven|twelve|thirteen|fourteen|fifteen|sixteen|seventeen|eighteen|nineteen|'
35
+ r'twenty|thirty|forty|fifty|sixty|seventy|eighty|ninety|'
36
+ r'hundred|thousand|million|billion|trillion|quadrillion|quintillion|'
37
+ r'first|second|third|fourth|fifth|sixth|seventh|eighth|ninth|tenth|'
38
+ r'eleventh|twelfth|thirteenth|fourteenth|fifteenth|sixteenth|seventeenth|'
39
+ r'eighteenth|nineteenth|twentieth|thirtieth|fortieth|fiftieth|sixtieth|'
40
+ r'seventieth|eightieth|ninetieth|hundredth|thousandth|millionth|billionth|'
41
+ r'trillionth|quadrillionth|quintillionth'
42
+ )
43
+ # # Matches entire spelled-out numbers, allowing hyphenated and space-separated formats.
44
+ # # Also accommodates "and" usage within numbers (e.g., "one hundred and twenty").
45
+ _NUM_WORDS_RE = (
46
+ r'\b(?:'
47
+ + _NUM_ORDINAL_WRDS_RE
48
+ + r')(?:[-\s]+(?:and[-\s]+)?(?:'
49
+ + _NUM_ORDINAL_WRDS_RE
50
+ + r'))*\b'
51
+ )
52
+
53
+ # Basic unit digits (1-9) as words.
54
+ _UNIT_DIGITS_WORDS = ["one", "two", "three", "four", "five", "six", "seven", "eight", "nine"]
55
+
56
+ # Multiples of ten (10-90) as words.
57
+ _TENS_MULTIPLES_WORDS = ["ten", "twenty", "thirty", "forty", "fifty", "sixty", "seventy", "eighty", "ninety"]
58
+
59
+ # Special cases for numbers between 11 and 19.
60
+ _TEEN_NUMERALS_WORDS = ["eleven", "twelve", "thirteen", "fourteen", "fifteen", "sixteen", "seventeen", "eighteen", "nineteen"]
61
+
62
+ # Maps ordinal words (e.g., "first", "second", "twentieth") to their corresponding numeric values.
63
+ _ORDINAL_MAPPING = {
64
+ "first": 1, "second": 2, "third": 3, "fourth": 4, "fifth": 5,
65
+ "sixth": 6, "seventh": 7, "eighth": 8, "ninth": 9, "tenth": 10,
66
+ "eleventh": 11, "twelfth": 12, "thirteenth": 13, "fourteenth": 14,
67
+ "fifteenth": 15, "sixteenth": 16, "seventeenth": 17, "eighteenth": 18,
68
+ "nineteenth": 19, "twentieth": 20, "thirtieth": 30, "fortieth": 40,
69
+ "fiftieth": 50, "sixtieth": 60, "seventieth": 70, "eightieth": 80,
70
+ "ninetieth": 90, "hundredth": 100, "thousandth": 1000,
71
+ "millionth": 1000000, "billionth": 1000000000,
72
+ "trillionth": 1000000000000,
73
+ "quadrillionth": 1000000000000000,
74
+ "quintillionth": 1000000000000000000,
75
+ }
76
+
77
+ # Dictionary that maps spelled-out ordinal words to their corresponding cardinal form and ordinal suffix.
78
+ _WORD_BASED_PATTERNS_RE = {
79
+ r'first$': ('one', 'st'),
80
+ r'second$': ('two', 'nd'),
81
+ r'third$': ('three', 'rd'),
82
+ r'fourth$': ('four', 'th'),
83
+ r'fifth$': ('five', 'th'),
84
+ r'sixth$': ('six', 'th'),
85
+ r'seventh$': ('seven', 'th'),
86
+ r'eighth$': ('eight', 'th'),
87
+ r'ninth$': ('nine', 'th'),
88
+ r'tenth$': ('ten', 'th'),
89
+ r'eleventh$': ('eleven', 'th'),
90
+ r'twelfth$': ('twelve', 'th'),
91
+ r'thirteenth$': ('thirteen', 'th'),
92
+ r'fourteenth$': ('fourteen', 'th'),
93
+ r'fifteenth$': ('fifteen', 'th'),
94
+ r'sixteenth$': ('sixteen', 'th'),
95
+ r'seventeenth$': ('seventeen', 'th'),
96
+ r'eighteenth$': ('eighteen', 'th'),
97
+ r'nineteenth$': ('nineteen', 'th'),
98
+ r'twentieth$': ('twenty', 'th'),
99
+ r'thirtieth$': ('thirty', 'th'),
100
+ r'fortieth$': ('forty', 'th'),
101
+ r'fiftieth$': ('fifty', 'th'),
102
+ r'sixtieth$': ('sixty', 'th'),
103
+ r'seventieth$': ('seventy', 'th'),
104
+ r'eightieth$': ('eighty', 'th'),
105
+ r'ninetieth$': ('ninety', 'th'),
106
+ r'hundredth$': ('hundred', 'th'),
107
+ r'thousandth$': ('thousand', 'th'),
108
+ r'millionth$': ('million', 'th'),
109
+ r'billionth$': ('billion', 'th'),
110
+ r'trillionth$': ('trillion', 'th'),
111
+ r'quadrillionth$': ('quadrillion', 'th'),
112
+ r'quintillionth$': ('quintillion', 'th'),
113
+ }
114
+
115
+ # Mapping for number multipliers used to scale values.
116
+ _MULTIPLIERS = {
117
+ "hundred": 100,
118
+ "thousand": 1000,
119
+ "million": 1000000,
120
+ "billion": 1000000000,
121
+ "trillion": 1000000000000,
122
+ "quadrillion": 1000000000000000,
123
+ "quintillion": 1000000000000000000,
124
+ }
125
+
126
+ # Maps individual Roman numeral symbols to their corresponding integer values based on the standard numeral system.
127
+ _ROMAN_NUMERAL_MAPPING = {
128
+ 'I': 1, 'V': 5, 'X': 10, 'L': 50,
129
+ 'C': 100, 'D': 500, 'M': 1000
130
+ }
131
+
132
+ # This regex pattern ensures strict validation of Roman numerals.
133
+ # It follows the standard Roman numeral rules and supports numbers from 1 (I) to 3999 (MMMCMXCIX).
134
+ # The pattern is divided into four groups:
135
+ # - (M{0,3}) → Matches 0 to 3 occurrences of 'M' (1000s place).
136
+ # - (CM|CD|D?C{0,3}) → Matches 900 (CM), 400 (CD), or up to 3 'C' after optional 'D' (100s place).
137
+ # - (XC|XL|L?X{0,3}) → Matches 90 (XC), 40 (XL), or up to 3 'X' after optional 'L' (10s place).
138
+ # - (IX|IV|V?I{0,3}) → Matches 9 (IX), 4 (IV), or up to 3 'I' after optional 'V' (1s place).
139
+ _ROMAN_NUMERAL_RE = r"^(M{0,3})(CM|CD|D?C{0,3})(XC|XL|L?X{0,3})(IX|IV|V?I{0,3})$"
140
+
141
+ # List of common currency symbols from around the world
142
+ _CURRENCY_SYMBOLS = [
143
+ r"$", # US Dollar
144
+ r"€", # Euro
145
+ r"£", # British Pound Sterling
146
+ r"¥", # Japanese Yen / Chinese Yuan
147
+ r"₹", # Indian Rupee
148
+ r"₩", # South Korean Won
149
+ r"₽", # Russian Ruble
150
+ r"R$", # Brazilian Real
151
+ r"₺", # Turkish Lira
152
+ r"฿", # Thai Baht
153
+ r"₫", # Vietnamese Dong
154
+ r"₱", # Philippine Peso
155
+ r"₴", # Ukrainian Hryvnia
156
+ r"₸", # Kazakhstani Tenge
157
+ r"֏", # Armenian Dram
158
+ r"₦", # Nigerian Naira
159
+ r"₵", # Ghanaian Cedi
160
+ r"Br", # Belarusian Ruble / Ethiopian Birr
161
+ r"₾", # Georgian Lari
162
+ r"₪", # Israeli Shekel
163
+ r"R", # South African Rand
164
+ r"HK$", # Hong Kong Dollar
165
+ r"S$", # Singapore Dollar
166
+ r"RM", # Malaysian Ringgit
167
+ r"Rp", # Indonesian Rupiah
168
+ r"Kč", # Czech Koruna
169
+ r"zł", # Polish Zloty
170
+ r"kr", # Scandinavian Krone (Denmark, Norway, Sweden)
171
+ r"Ft", # Hungarian Forint
172
+ r"lei", # Romanian Leu
173
+ r"лв", # Bulgarian Lev
174
+ r"дин", # Serbian Dinar
175
+ r"kn", # Croatian Kuna
176
+ r"ден", # Macedonian Denar
177
+ r"L", # Albanian Lek
178
+ r"Br", # Repeated for Belarusian Ruble / Ethiopian Birr
179
+ r"S/.", # Peruvian Sol
180
+ r"CHF" # Swiss Franc
181
+ ]
182
+
183
+ ## Helper Functions
184
+ ##━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
185
+
186
+
187
+ # String Parsing & Tokenization
188
+ #────────────────────────────────────────────────────────────────────────────
189
+ def __parseNumericToken(s: str, first_only: bool = True, wrap_single: bool = False):
190
+ """
191
+ Finds spelled-out or digit-based tokens in a string and returns
192
+ either the first match or all.
193
+
194
+ This function searches for numeric tokens in two ways:
195
+ 1. Digit-based (e.g., '42', '1,234', '2.50')
196
+ 2. Spelled-out (e.g., 'twenty-five', 'third')
197
+ """
198
+ def _tokenize_digit(text: str, first_only: bool = True):
199
+ """
200
+ Finds digit-based tokens in a string, including optional decimals, and commas.
201
+ """
202
+ pattern = r'-?\d+(?:,\d{3})*(?:\.\d+)?'
203
+
204
+ matches = list(re.finditer(pattern, text))
205
+ if not matches:
206
+ return None if first_only else []
207
+
208
+ # Convert each match to a string, removing commas
209
+ found = [re.sub(r',', '', m.group(0)) for m in matches]
210
+
211
+ if first_only:
212
+ # Return a single-element list
213
+ return [found[0]]
214
+ else:
215
+ return found
216
+
217
+ def _tokenize_alpha(text: str, first_only: bool = True):
218
+ """
219
+ Finds spelled-out numbers or ordinals in a string, using _NUM_WORDS_RE.
220
+ """
221
+ matches = list(re.finditer(_NUM_WORDS_RE, text, re.IGNORECASE))
222
+ if not matches:
223
+ return None if first_only else []
224
+
225
+ # Extract each spelled-out token
226
+ found = [text[m.start():m.end()] for m in matches]
227
+
228
+ if first_only:
229
+ return [found[0]]
230
+ else:
231
+ return found
232
+
233
+ if not s:
234
+ return None if first_only else []
235
+
236
+ # Get all digit-based tokens
237
+ digit_tokens = _tokenize_digit(s, first_only=False) # get all
238
+ # Get all spelled-out tokens
239
+ alpha_tokens = _tokenize_alpha(s, first_only=False) # get all
240
+
241
+ # Combine results
242
+ combined = []
243
+ if digit_tokens:
244
+ combined += digit_tokens
245
+ if alpha_tokens:
246
+ combined += alpha_tokens
247
+
248
+ if not combined:
249
+ return None if first_only else []
250
+
251
+ # If user wants only the first match, return the first item
252
+ if first_only:
253
+ result = combined[0]
254
+ return [result] if wrap_single else result # Wrap in list if requested
255
+ else:
256
+ return combined
257
+
258
+ def __parseRomanNumeral(s: str):
259
+ """
260
+ Parses a Roman numeral string and converts it into an integer.
261
+
262
+ This function ensures the Roman numeral follows valid ordering rules and
263
+ correctly applies subtractive notation.
264
+ """
265
+ if not s or not re.match(_ROMAN_NUMERAL_RE, s):
266
+ return None
267
+
268
+ num, prev = 0, 0
269
+
270
+ # Convert from right to left
271
+ for char in reversed(s):
272
+ curr = _ROMAN_NUMERAL_MAPPING[char]
273
+ num = num - curr if curr < prev else num + curr
274
+ prev = curr
275
+
276
+ return num if num > 0 else None # Ensure non-zero positive value
277
+
278
+ def __validate_numstr(n: Union[int, float, str], clean: bool = False) -> Optional[str]:
279
+ """
280
+ Validates whether the input contains a valid numeric value after removing
281
+ currency symbols and commas.
282
+
283
+ This function converts the input to a string, removes known currency symbols
284
+ and commas, and checks if the remaining content is a valid numeric value.
285
+ """
286
+ numstr = str(n)
287
+ clean_numstr = reduce(lambda s, sign: s.replace(sign, ""), _CURRENCY_SYMBOLS, numstr).replace(",", "")
288
+ clean_numstr = clean_numstr.replace("+", "").replace("-", "")
289
+ # Check if the string is a valid number
290
+ try:
291
+ float(clean_numstr) # Check for validity
292
+ except ValueError:
293
+ return None
294
+ return clean_numstr if clean else n
295
+
296
+ # Boolean & Utility Functions
297
+ #────────────────────────────────────────────────────────────────────────────
298
+ def __switch(x):
299
+ """
300
+ Toggle a boolean value.
301
+
302
+ Parameters:
303
+ ──────────────────────────
304
+ x (bool): The boolean value to switch.
305
+
306
+ Returns:
307
+ ──────────────────────────
308
+ bool: The opposite of the input boolean value.
309
+ """
310
+ return not x
311
+
312
+
313
+
314
+
315
+ ## General Formatting & Numeric Processing
316
+ ##━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
317
+
318
+ # These are intermediate utility functions for number formatting
319
+ #────────────────────────────────────────────────────────────────────────────
320
+ def insertSep(n: Union[int, float, str], sep: str = ",") -> Optional[str]:
321
+ """
322
+ Formats a number by inserting a specified thousands separator while preserving
323
+ any non-numeric characters (e.g., currency symbols, quotes).
324
+
325
+ This function extracts the numeric portion of the input, formats it with
326
+ the specified thousands separator, and then restores the original non-numeric
327
+ characters at the beginning and end.
328
+
329
+ Parameters:
330
+ ──────────────────────────
331
+ n (str, int, or float): The input number, which may contain non-numeric characters.
332
+ sep (str, optional): The separator to use for thousands grouping. Default is a comma (",").
333
+
334
+ Returns:
335
+ ──────────────────────────
336
+ str: The formatted number with thousands separators, retaining any original non-numeric characters.
337
+ None: If the input does not contain a valid number.
338
+ """
339
+ num_str = str(n) # Ensure the input is a string
340
+ num_str = " ".join(num_str.split())
341
+
342
+ # Extract non-numeric characters at the beginning and end
343
+ match = re.match(r"(^\D*)(?:[\d,.]+)?(\D*$)", num_str)
344
+ if not match:
345
+ return None
346
+
347
+ prefix, suffix = match.groups()
348
+
349
+ # Check if the string is a valid number
350
+ if not __validate_numstr(num_str):
351
+ return None
352
+ else:
353
+ cleaned_numeric_part = __validate_numstr(num_str, True)
354
+
355
+ # Determine if the number contains a decimal
356
+ if '.' in cleaned_numeric_part:
357
+ integer_part, decimal_part = cleaned_numeric_part.split('.')
358
+ integer_part = int(integer_part) # Convert to int to remove leading zeros if any
359
+ else:
360
+ integer_part, decimal_part = int(cleaned_numeric_part), None
361
+
362
+ # Format the integer part with the specified separator
363
+ formatted_integer_part = f"{integer_part:,}".replace(",", sep)
364
+
365
+ # Reassemble the formatted number with the original non-numeric characters
366
+ if decimal_part is not None:
367
+ formatted_number = f"{formatted_integer_part}.{decimal_part}"
368
+ else:
369
+ formatted_number = formatted_integer_part
370
+
371
+ return f"{prefix}{formatted_number}{suffix}"
372
+
373
+ def formatDecimal(n: Union[int, float, str], place: int = 5) -> Optional[str]:
374
+ """
375
+ Formats a given number to ensure it has a fixed number of decimal places.
376
+
377
+ Parameters:
378
+ ──────────────────────────
379
+ n (Union[int, float, str]): The input number as a string, integer, or float.
380
+ place (int, optional): The number of decimal places to enforce. Default is **5**.
381
+
382
+ Returns:
383
+ ──────────────────────────
384
+ Optional[str]: The formatted number with the specified decimal places.
385
+ Returns **None** if the input is invalid.
386
+ """
387
+ # Ensure the input is a string
388
+ n = str(n)
389
+ n = " ".join(n.split())
390
+
391
+ # Check if the string is a valid number
392
+ if not __validate_numstr(n):
393
+ return None
394
+
395
+ place = int(place)
396
+
397
+ if '.' not in n:
398
+ n = f'{n}.{str(10**place).replace("1", "")}'
399
+
400
+ integer_part, decimal_part = n.split('.')
401
+ if len(decimal_part) < place:
402
+ decimal_part = decimal_part.ljust(place, '0')
403
+ elif len(decimal_part) > place:
404
+ decimal_part = decimal_part[:place]
405
+
406
+ return f"{integer_part}.{decimal_part}"
407
+
408
+
409
+
410
+
411
+ ## Main Conversion Logic
412
+ ##━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
413
+
414
+
415
+ # WORD-BASED NUMBER TO INTEGER: CONVERT CARDINAL NUMBERS IN WORD FORM (E.G., "TWENTY-FIVE") TO INTEGERS
416
+ #───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
417
+ def wordsToInt(s: str, thousands_sep: bool = False, sep: str = ","):
418
+ """
419
+ Converts a spelled-out number in English into its integer equivalent.
420
+
421
+ This function processes cardinal numbers written in words and returns their
422
+ corresponding integer values. It supports numbers up to quintillions, including
423
+ multi-word formats and hyphenated numbers (e.g., "twenty-one"). Negative numbers
424
+ are also recognized if prefixed with "negative" or "minus".
425
+
426
+ The function can optionally format the output with thousands separators for better readability.
427
+
428
+ Parameters:
429
+ ──────────────────────────
430
+ s (str): A string representing a number in English (e.g., "two hundred and fifty-six").
431
+ thousands_sep (bool, optional): If True, formats the output with a separator (default is False).
432
+ sep (str, optional): The character used as a thousands separator when `thousands_sep` is True (default is ',').
433
+
434
+ Returns:
435
+ ──────────────────────────
436
+ int or None: The integer representation of the input if successfully parsed, otherwise None.
437
+
438
+ Notes:
439
+ ──────────────────────────
440
+ - Does not process ordinal numbers (e.g., "first", "twenty-first").
441
+ - Assumes grammatically correct number formatting.
442
+ - Ignores the word "and" as it does not affect numerical value (e.g., "one hundred and five" = "one hundred five").
443
+ - Recognizes and processes compound words (e.g., "forty-two", "ninety-nine").
444
+ - If an unrecognized word is found, the function returns `None`.
445
+ """
446
+ if not s:
447
+ return None
448
+
449
+ # Detect negative at the front
450
+ is_negative = False
451
+ s = s.strip().lower()
452
+ if s.startswith("negative "):
453
+ is_negative = True
454
+ s = s.replace("negative ", "", 1)
455
+ elif s.startswith("minus "):
456
+ is_negative = True
457
+ s = s.replace("minus ", "", 1)
458
+
459
+ # If it's an ordinal phrase, we skip it by returning None
460
+ token = __parseNumericToken(s)
461
+ if token and ordinalWordsToInt(token):
462
+ return None
463
+
464
+ units_dict = {word: i + 1 for i, word in enumerate(_UNIT_DIGITS_WORDS)}
465
+ tens_dict = {word: (i + 1) * 10 for i, word in enumerate(_TENS_MULTIPLES_WORDS)}
466
+ teens_dict = {word: i + 11 for i, word in enumerate(_TEEN_NUMERALS_WORDS)}
467
+
468
+ # Pre-compute some compound forms (e.g. "twenty-hundred" though not standard, etc.)
469
+ for mult in _MULTIPLIERS:
470
+ for word_list in [_UNIT_DIGITS_WORDS, _TENS_MULTIPLES_WORDS]:
471
+ for w in word_list:
472
+ compound = w + "-" + mult
473
+ if w in tens_dict:
474
+ units_dict[compound] = tens_dict[w] * _MULTIPLIERS[mult]
475
+ else:
476
+ units_dict[compound] = units_dict[w] * _MULTIPLIERS[mult]
477
+
478
+ # Split out hyphens/spaces, skip "and" tokens
479
+ words_list = re.split(r"[\s-]+", s)
480
+ words_list = [w for w in words_list if w != "and"] # skip "and"
481
+
482
+ number = 0
483
+ temp_number = 0
484
+
485
+ for word in words_list:
486
+ if word in units_dict:
487
+ temp_number += units_dict[word]
488
+ elif word in teens_dict:
489
+ temp_number += teens_dict[word]
490
+ elif word in tens_dict:
491
+ temp_number += tens_dict[word]
492
+ elif word in _MULTIPLIERS:
493
+ # For 'hundred', multiply the existing temp_number by 100
494
+ # For thousand/million/etc., multiply and then "commit" to number
495
+ temp_number *= _MULTIPLIERS[word]
496
+ if _MULTIPLIERS[word] >= 1000:
497
+ number += temp_number
498
+ temp_number = 0
499
+ else:
500
+ # If unknown word, fail
501
+ return None
502
+
503
+ number += temp_number
504
+
505
+ if number == 0:
506
+ return None
507
+
508
+ if is_negative:
509
+ number = -number
510
+ if thousands_sep:
511
+ number = insertSep(number, sep=sep)
512
+ return number
513
+
514
+
515
+ # WORD-BASED NUMBER TO INTEGER: CONVERT ORDINAL NUMBERS IN WORD FORM (E.G., "TWENTY-FIRST") TO INTEGERS
516
+ #───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
517
+ def ordinalWordsToInt(s: str, to_num: bool = False, thousands_sep: bool = False, sep: str = ","):
518
+ """
519
+ Converts a spelled-out ordinal number in English to its numeric form.
520
+
521
+ This function processes ordinal words and converts them into either a numeric
522
+ string with a suffix (e.g., "1st", "21st") or an integer (if `to_num=True`).
523
+ It handles multi-word ordinals, hyphenated ordinals, and recognizes negative
524
+ ordinal numbers when prefixed with "negative" or "minus".
525
+
526
+ Parameters:
527
+ ──────────────────────────
528
+ s (str): A string representing an ordinal number in English
529
+ (e.g., "twenty-first", "hundredth").
530
+ to_num (bool, optional): If True, returns the ordinal as an integer instead
531
+ of a string with a suffix. Default is False.
532
+ thousands_sep (bool, optional): If True, formats the output with a separator (default is False).
533
+ sep (str, optional): The character used as a thousands separator when `thousands_sep` is True (default is ',').
534
+
535
+ Returns:
536
+ ──────────────────────────
537
+ str or int or None: The numeric ordinal representation as a string with a suffix
538
+ (e.g., "21st") or as an integer (if `to_num=True`), or
539
+ None if parsing fails.
540
+ """
541
+ # Define lists of spelled-out cardinal and ordinal numbers and combine both lists
542
+ cardinal_numbers = ['zero', *_UNIT_DIGITS_WORDS, *_TEEN_NUMERALS_WORDS, *_TENS_MULTIPLES_WORDS]
543
+ ordinal_numbers = list(_ORDINAL_MAPPING.keys())
544
+ number_words = cardinal_numbers + ordinal_numbers
545
+
546
+ # Build a regex pattern that matches a hyphen only if it is between two valid number words
547
+ pattern = r'\b(' + '|'.join(number_words) + r')-(' + '|'.join(number_words) + r')\b'
548
+
549
+ # s = re.sub(r'(\w)-(\w)', r'\1 \2', s)
550
+
551
+ # Replaces hyphens only if they are between spelled-out cardinal or ordinal number words.
552
+ s = re.sub(pattern, r'\1 \2', s, flags=re.IGNORECASE)
553
+
554
+ if not s:
555
+ return None
556
+
557
+ # Detect negative at the front (rare for ordinals, but let's allow it)
558
+ is_negative = False
559
+ s = s.strip().lower()
560
+ if s.startswith("negative "):
561
+ is_negative = True
562
+ s = s.replace("negative ", "", 1)
563
+ elif s.startswith("minus "):
564
+ is_negative = True
565
+ s = s.replace("minus ", "", 1)
566
+
567
+ def _strip_num(ordinal_number):
568
+ """
569
+ From '21st', '32nd', etc. -> integer
570
+ """
571
+ ordinal_match = re.match(r'^-?(\d+)(st|nd|rd|th)$', ordinal_number)
572
+ return int(ordinal_match.group(1)) if ordinal_match else None
573
+
574
+ def _simple_ord(tok):
575
+ """
576
+ Simple single-word ordinal: 'first' -> '1st', 'second' -> '2nd', etc.
577
+ """
578
+ # Attempt direct mapping
579
+ if tok in _ORDINAL_MAPPING:
580
+ base_val = _ORDINAL_MAPPING[tok]
581
+ suffix = ordinalSuffix(base_val) # "st", "nd", "rd", "th"
582
+ # Return with or without numeric form
583
+ if to_num:
584
+ return -base_val if is_negative else base_val
585
+ else:
586
+ return f"{'-' if is_negative else ''}{base_val}{suffix}"
587
+ return None
588
+
589
+ def _complex_ord(tok):
590
+ """
591
+ Handle multi-word ordinals: 'twenty first' -> '21st', 'one hundred and first' -> '101st'
592
+ """
593
+ # We'll try removing the last word as an ordinal ending, then parse the front as cardinal
594
+ last_word_match = re.search(r'\b(\w+)\b$', tok)
595
+ if not last_word_match:
596
+ return None
597
+ last_word = last_word_match.group()
598
+
599
+ # Everything except the last word
600
+ front_string = re.sub(r'\s*\b\w+\b$', '', tok).strip()
601
+
602
+ # Convert front part to integer (cardinal)
603
+ front_number = wordsToInt(front_string)
604
+
605
+ if front_number is None:
606
+ return None
607
+
608
+ # Then interpret the last word as a single-word ordinal
609
+ last_ordinal_num = _simple_ord(last_word)
610
+ if last_ordinal_num is None:
611
+ return None
612
+
613
+ # If last_ordinal_num is integer (to_num=True) or a string with suffix
614
+ if isinstance(last_ordinal_num, int):
615
+ # If it is an integer, just sum
616
+ complete = front_number + last_ordinal_num
617
+ return complete if not is_negative else -complete
618
+ else:
619
+ # Otherwise it's something like "21st"
620
+ # Extract the integer portion from "21st"
621
+ stripped_val = _strip_num(last_ordinal_num)
622
+ if stripped_val is None:
623
+ return None
624
+ complete = front_number + stripped_val
625
+ final_suffix = ordinalSuffix(complete)
626
+ sign = "-" if is_negative else ""
627
+ return f"{sign}{complete}{final_suffix}"
628
+
629
+ # Parse to check if single or multiple words
630
+ # We'll use parseNumericToken to see if there's a direct token
631
+ # then handle single vs multi logic
632
+ token = __parseNumericToken(s)
633
+ if not token:
634
+ return None
635
+
636
+ # If more than one word, attempt complex
637
+ if len(token.split()) > 1:
638
+ check_number = _complex_ord(token)
639
+ if check_number is not None:
640
+ if to_num and thousands_sep:
641
+ return insertSep(check_number, sep=sep)
642
+ else:
643
+ return check_number
644
+ else:
645
+ # single word
646
+ check_number = _simple_ord(token)
647
+ if check_number is not None:
648
+ if to_num and thousands_sep:
649
+ return insertSep(check_number, sep=sep)
650
+ else:
651
+ return check_number
652
+ return None
653
+
654
+ # WORD-BASED NUMBER TO INTEGER: CONVERT NUMERIC STRINGS (E.G., "1,234", "42ND") TO INTEGERS
655
+ #───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
656
+ def stringToInt(s: str, to_str: bool = False, thousands_sep: bool = False, sep: str = ","):
657
+ """
658
+ Converts a numeric string into an integer or string representation.
659
+
660
+ This function processes numeric strings containing digits, optionally formatted
661
+ with commas (e.g., "1,234"), and ordinal suffixes (e.g., "2nd"). It also handles
662
+ negative numbers prefixed with "negative" or "minus". The function returns an
663
+ integer by default but can return a string representation if `to_str=True`.
664
+
665
+ Parameters:
666
+ ──────────────────────────
667
+ s (str): A numeric string, potentially containing commas or ordinal suffixes.
668
+ to_str (bool, optional): If True, returns the result as a string instead of an integer.
669
+ Default is False.
670
+ thousands_sep (bool, optional): If True, formats the output with a separator (default is False).
671
+ sep (str, optional): The character used as a thousands separator when `thousands_sep` is True (default is ',').
672
+
673
+ Returns:
674
+ ──────────────────────────
675
+ int or str or None: The parsed integer or its string representation (if `to_str=True`).
676
+ Returns None if parsing fails.
677
+ """
678
+ if not s:
679
+ return None
680
+
681
+ # Detect negative at the front
682
+ is_negative = False
683
+ number_str = s.strip()
684
+ if number_str.startswith("-"):
685
+ is_negative = True
686
+ number_str = number_str[1:].strip()
687
+ elif number_str.lower().startswith("negative "):
688
+ is_negative = True
689
+ number_str = number_str.lower().replace("negative ", "", 1)
690
+ elif number_str.lower().startswith("minus "):
691
+ is_negative = True
692
+ number_str = number_str.lower().replace("minus ", "", 1)
693
+
694
+ tokens = __parseNumericToken(number_str)
695
+ # if tokens and re.match(r'^-?\d+$', str(tokens)):
696
+ # val = int(tokens)
697
+ # val = -val if is_negative else val
698
+ # return str(val) if to_str else val
699
+ if tokens and re.match(r'^-?\d+$', str(tokens)):
700
+ val = int(tokens)
701
+ val = -val if is_negative else val
702
+ if thousands_sep and to_str:
703
+ return insertSep(val, sep=sep)
704
+ return str(val) if to_str else val
705
+ return None
706
+
707
+
708
+
709
+
710
+ # INTEGER TO WORD-BASED NUMBER: CONVERT INTEGER VALUES (E.G., 256) TO CARDINAL NUMBERS IN WORD FORM (E.G., "TWO HUNDRED FIFTY-SIX")
711
+ #───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
712
+ def intToWords(n: Union[int, float], thousands_sep: bool = False):
713
+ """
714
+ Converts an integer or float into its spelled-out English words representation.
715
+
716
+ This function transforms numerical values into their corresponding English
717
+ words. It supports both positive and negative numbers, including large values
718
+ up to quintillions. Additionally, it can handle floating-point numbers by
719
+ spelling out both the integer and decimal portions separately.
720
+
721
+ If `thousands_sep=True`, large segments (e.g., thousands, millions, billions)
722
+ are separated by commas in the output.
723
+
724
+ Parameters:
725
+ ──────────────────────────
726
+ n (int or float): The number to be converted to words.
727
+ thousands_sep (bool, optional): If True, inserts commas between large segments
728
+ for better readability. Default is False.
729
+
730
+ Returns:
731
+ ──────────────────────────
732
+ str or None: The spelled-out English representation of the number, or None
733
+ if input is invalid.
734
+ """
735
+ def _from_int(x):
736
+ if x == 0:
737
+ return "zero"
738
+
739
+ def one(num):
740
+ switcher = {
741
+ 1: 'one', 2: 'two', 3: 'three', 4: 'four', 5: 'five',
742
+ 6: 'six', 7: 'seven', 8: 'eight', 9: 'nine'
743
+ }
744
+ return switcher.get(num, '')
745
+
746
+ def two_less_20(num):
747
+ switcher = {
748
+ 10: 'ten', 11: 'eleven', 12: 'twelve', 13: 'thirteen', 14: 'fourteen',
749
+ 15: 'fifteen', 16: 'sixteen', 17: 'seventeen', 18: 'eighteen', 19: 'nineteen'
750
+ }
751
+ return switcher.get(num, '')
752
+
753
+ def ten(num):
754
+ switcher = {
755
+ 2: 'twenty', 3: 'thirty', 4: 'forty', 5: 'fifty',
756
+ 6: 'sixty', 7: 'seventy', 8: 'eighty', 9: 'ninety'
757
+ }
758
+ return switcher.get(num, '')
759
+
760
+ def two(num):
761
+ if not num:
762
+ return ''
763
+ elif num < 10:
764
+ return one(num)
765
+ elif num < 20:
766
+ return two_less_20(num)
767
+ else:
768
+ tenner = num // 10
769
+ rest = num % 10
770
+ return ten(tenner) + ('-' + one(rest) if rest else '')
771
+
772
+ def three(num):
773
+ hundred = num // 100
774
+ rest = num % 100
775
+ if hundred and rest:
776
+ return one(hundred) + ' hundred ' + two(rest)
777
+ elif hundred and not rest:
778
+ return one(hundred) + ' hundred'
779
+ else:
780
+ return two(rest)
781
+
782
+ # Break the number into billions, millions, thousands, and the remainder
783
+ # Extended to quintillions
784
+ # We'll do repeated modulus and division:
785
+ # e.g. 1,234,567,890,123 -> segments for trillions, billions, millions, thousands, rest
786
+ # We'll store all segments in ascending order, then build from largest to smallest for readability.
787
+ # For now, let's just go up to quintillions.
788
+ abs_num = abs(x)
789
+
790
+ quintillion = abs_num // 1000000000000000000
791
+ remainder_q = abs_num % 1000000000000000000
792
+
793
+ quadrillion = remainder_q // 1000000000000000
794
+ remainder_quad = remainder_q % 1000000000000000
795
+
796
+ trillion = remainder_quad // 1000000000000
797
+ remainder_tril = remainder_quad % 1000000000000
798
+
799
+ billion = remainder_tril // 1000000000
800
+ remainder_bill = remainder_tril % 1000000000
801
+
802
+ million = remainder_bill // 1000000
803
+ remainder_mill = remainder_bill % 1000000
804
+
805
+ thousand = remainder_mill // 1000
806
+ remainder = remainder_mill % 1000
807
+
808
+ segments = []
809
+ if quintillion:
810
+ segments.append(three(quintillion) + " quintillion")
811
+ if quadrillion:
812
+ segments.append(three(quadrillion) + " quadrillion")
813
+ if trillion:
814
+ segments.append(three(trillion) + " trillion")
815
+ if billion:
816
+ segments.append(three(billion) + " billion")
817
+ if million:
818
+ segments.append(three(million) + " million")
819
+ if thousand:
820
+ segments.append(three(thousand) + " thousand")
821
+ if remainder:
822
+ segments.append(three(remainder))
823
+
824
+ result = ""
825
+ if segments:
826
+ if thousands_sep and len(segments) > 1:
827
+ result = ", ".join(seg for seg in segments if seg).strip()
828
+ else:
829
+ result = " ".join(seg for seg in segments if seg).strip()
830
+ else:
831
+ result = "zero"
832
+
833
+ # Attach negative sign if needed
834
+ if x < 0:
835
+ result = "negative " + result
836
+
837
+ return result
838
+
839
+ def _from_float(num):
840
+ """
841
+ Very basic approach to handle floats by splitting at the decimal.
842
+ """
843
+ whole_str, decimal_str = str(num).split(".")
844
+ whole_part = _from_int(int(whole_str))
845
+ # Convert each digit in the decimal part to words, or parse the entire decimal as an integer:
846
+ # "45" -> "forty-five" or "four five"
847
+ dec_int = int(decimal_str)
848
+ decimal_words = _from_int(dec_int)
849
+
850
+ return f"{whole_part} point {decimal_words}"
851
+
852
+ n_str = str(n)
853
+ if "." in n_str:
854
+ try:
855
+ float_val = float(n_str)
856
+ return _from_float(float_val)
857
+ except ValueError:
858
+ return None
859
+ else:
860
+ try:
861
+ int_val = int(n_str)
862
+ return _from_int(int_val)
863
+ except ValueError:
864
+ return None
865
+
866
+
867
+ # INTEGER TO WORD-BASED NUMBER: CONVERT INTEGER VALUES (E.G., 21) TO ORDINAL NUMBERS IN WORD FORM (E.G., "TWENTY-FIRST")
868
+ #───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
869
+ def intToOrdinalWords(n: int):
870
+ """
871
+ Converts an integer into its ordinal spelled-out English form.
872
+
873
+ This function takes an integer and returns its ordinal representation in words.
874
+ It correctly handles standard English transformations for ordinal numbers,
875
+ including irregular forms like "first", "second", "third", "twelfth", and
876
+ suffix changes for numbers ending in "-y" (e.g., "twenty" -> "twentieth").
877
+
878
+ The function supports both positive and negative numbers.
879
+
880
+ Parameters:
881
+ ──────────────────────────
882
+ n (int): The integer to be converted into ordinal words.
883
+
884
+ Returns:
885
+ ──────────────────────────
886
+ str or None: The ordinal representation of the number as a string,
887
+ or None if conversion fails.
888
+ """
889
+ words = intToWords(n, thousands_sep=False)
890
+ if not words:
891
+ return None
892
+
893
+ # We'll split the spelled-out form, then transform the last word.
894
+ # Note: This is a simplistic approach and might need special handling for multi-segment final words.
895
+ word_parts = words.split()
896
+ if not word_parts:
897
+ return None
898
+
899
+ # Identify the negative sign if it exists
900
+ is_negative = False
901
+ if word_parts[0] == "negative":
902
+ is_negative = True
903
+ word_parts = word_parts[1:] # remove "negative"
904
+
905
+ last_word = word_parts[-1]
906
+
907
+ def _replace_end(full_word, old_end, new_end):
908
+ return full_word[: -len(old_end)] + new_end if full_word.endswith(old_end) else full_word
909
+
910
+ # We'll do a set of special transformations:
911
+ # one -> first, two -> second, three -> third, etc.
912
+ # This is a subset of patterns.
913
+ if last_word.endswith("one"):
914
+ word_parts[-1] = _replace_end(last_word, "one", "first")
915
+ elif last_word.endswith("two"):
916
+ word_parts[-1] = _replace_end(last_word, "two", "second")
917
+ elif last_word.endswith("three"):
918
+ word_parts[-1] = _replace_end(last_word, "three", "third")
919
+ elif last_word.endswith("five"):
920
+ word_parts[-1] = _replace_end(last_word, "five", "fifth")
921
+ elif last_word.endswith("eight"):
922
+ word_parts[-1] = _replace_end(last_word, "eight", "eighth")
923
+ elif last_word.endswith("nine"):
924
+ word_parts[-1] = _replace_end(last_word, "nine", "ninth")
925
+ elif last_word.endswith("twelve"):
926
+ word_parts[-1] = _replace_end(last_word, "twelve", "twelfth")
927
+ elif last_word.endswith("y"):
928
+ word_parts[-1] = _replace_end(last_word, "y", "ieth") # e.g. "twenty" -> "twentieth", "thirty" -> "thirtieth"
929
+ elif last_word.endswith("teen"):
930
+ word_parts[-1] = _replace_end(last_word, "teen", "teenth") # e.g. "fourteen" -> "fourteenth"
931
+ else:
932
+ word_parts[-1] = word_parts[-1] + "th" # Generic
933
+
934
+ if is_negative:
935
+ return "negative " + " ".join(word_parts)
936
+ else:
937
+ return " ".join(word_parts)
938
+
939
+
940
+
941
+
942
+ # ORDINAL NUMBER UTILITIES: EXTRACT THE APPROPRIATE SUFFIX ("ST", "ND", "RD", "TH") FOR AN INTEGER
943
+ #───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
944
+ def ordinalSuffix(n: int):
945
+ """
946
+ Determines the appropriate English ordinal suffix for a given integer.
947
+
948
+ This function returns the correct ordinal suffix ("st", "nd", "rd", "th")
949
+ based on standard English rules. It properly accounts for special cases
950
+ where numbers ending in 11, 12, or 13 always take "th".
951
+
952
+ Parameters:
953
+ ──────────────────────────
954
+ n (int): The integer for which to determine the ordinal suffix.
955
+
956
+ Returns:
957
+ ──────────────────────────
958
+ str: The appropriate ordinal suffix ('st', 'nd', 'rd', or 'th').
959
+ """
960
+ last_two = abs(n) % 100
961
+ last_digit = abs(n) % 10
962
+ if last_two in (11, 12, 13):
963
+ return "th"
964
+ else:
965
+ if last_digit == 1:
966
+ return "st"
967
+ elif last_digit == 2:
968
+ return "nd"
969
+ elif last_digit == 3:
970
+ return "rd"
971
+ else:
972
+ return "th"
973
+
974
+ # ORDINAL NUMBER UTILITIES: REMOVE THE ORDINAL ENDING FROM A WORD-BASED NUMBER (E.G., "TWENTIETH" → "TWENTY")
975
+ #───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
976
+ def stripOrdinalSuffix(s: str):
977
+ """
978
+ Strips the ordinal suffix from a spelled-out ordinal number, returning the base
979
+ cardinal form and the suffix separately.
980
+
981
+ This function identifies and removes ordinal suffixes from spelled-out ordinal
982
+ numbers (e.g., "twentieth" → "twenty", "twenty-first" → "twenty-one"). It returns
983
+ a tuple containing the base cardinal number as a string and the ordinal suffix.
984
+
985
+ Parameters:
986
+ ──────────────────────────
987
+ s (str): A spelled-out ordinal number (e.g., "seventh", "thirty-second").
988
+
989
+ Returns:
990
+ ──────────────────────────
991
+ tuple or None: A tuple containing the base cardinal number (str) and its
992
+ ordinal suffix (str), or None if no ordinal suffix is detected.
993
+ """
994
+ number_str = s
995
+ suffix = None
996
+ for pattern, (replacement, suffix_to_remove) in _WORD_BASED_PATTERNS_RE.items():
997
+ if re.search(pattern, s, flags=re.IGNORECASE):
998
+ suffix = suffix_to_remove
999
+ number_str = re.sub(pattern, replacement, number_str, flags=re.IGNORECASE)
1000
+ break # Stop at first match
1001
+ if suffix is None:
1002
+ return None
1003
+ return (number_str, suffix)
1004
+
1005
+
1006
+
1007
+
1008
+ # GENERAL NUMBER EXTRACTION: IDENTIFY AND CONVERT ALL NUMERIC VALUES (CARDINAL OR ORDINAL) FROM A STRING
1009
+ #───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
1010
+ def extractNumericValue(s: str, allnum: bool = True):
1011
+ """
1012
+ Extracts and converts all numeric values (cardinal or ordinal) from a string into a list of integers.
1013
+
1014
+ This function searches for multiple numeric values, whether they are:
1015
+ - Digit-based (e.g., "42", "1,234")
1016
+ - Ordinals (e.g., "third", "42nd")
1017
+ - Spelled-out numbers (e.g., "twenty-five")
1018
+
1019
+ If a negative indicator ("negative" or "minus") is present at the start of the string,
1020
+ it applies negativity only to the **first** detected number.
1021
+
1022
+ Parameters:
1023
+ ──────────────────────────
1024
+ s (str): A string potentially containing multiple numeric values.
1025
+ allnum (bool, optional): Determines whether to return **all** numeric matches or just the **first** one.
1026
+ - `True` (default): Returns a list of all numbers found in the string.
1027
+ - `False`: Returns only the first number found.
1028
+
1029
+ Returns:
1030
+ ──────────────────────────
1031
+ list[int] or None:
1032
+ - A list of parsed integers if numeric values are found.
1033
+ - None if no numbers are detected.
1034
+ """
1035
+ if not s:
1036
+ return None
1037
+
1038
+ # Handle negative at the start of the string
1039
+ is_negative = False
1040
+ string = s.strip().lower()
1041
+ if string.startswith("negative "):
1042
+ is_negative = True
1043
+ string = string.replace("negative ", "", 1)
1044
+ elif string.startswith("minus "):
1045
+ is_negative = True
1046
+ string = string.replace("minus ", "", 1)
1047
+
1048
+ # tokens = __parseNumericToken(s, first_only=False) # Get all matches
1049
+ tokens = __parseNumericToken(s, first_only=__switch(allnum), wrap_single=True) # Get all matches if allnum == True. The switch function swithes allnum boolen to False.
1050
+ if not tokens:
1051
+ return None
1052
+
1053
+ # if not isinstance(tokens, list):
1054
+ # tokens=[tokens]
1055
+
1056
+ def _check_and_return(num):
1057
+ return num if isinstance(num, int) else None
1058
+
1059
+ # Define functions to apply for conversion
1060
+ funcs_and_kwargs = [
1061
+ (ordinalWordsToInt, {'to_num': True}),
1062
+ (stringToInt, {'to_str': False}),
1063
+ (wordsToInt, {}),
1064
+ ]
1065
+
1066
+ # Process all tokens
1067
+ parsed_numbers = []
1068
+ for token in tokens:
1069
+ for func, kwargs in funcs_and_kwargs:
1070
+ result = func(token, **kwargs)
1071
+ number = _check_and_return(result)
1072
+ if number is not None:
1073
+ # Apply negativity only to the first detected number
1074
+ if is_negative and not parsed_numbers: # Only negate the **first** number found
1075
+ number = -abs(number)
1076
+ parsed_numbers.append(number)
1077
+ break # Stop trying other functions once parsed successfully
1078
+
1079
+ # Return single integer if only one number is found, otherwise return list
1080
+ if not parsed_numbers:
1081
+ return None
1082
+ return parsed_numbers[0] if len(parsed_numbers) == 1 else parsed_numbers
1083
+
1084
+
1085
+
1086
+ # ROMAN NUMERALS: CONVERT ROMAN NUMERALS (E.G., "XIV") TO INTEGERS
1087
+ #───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
1088
+ def romanToInt(s: str, to_str: bool = False):
1089
+ """
1090
+ Converts a Roman numeral string into its integer value.
1091
+
1092
+ This function parses and converts valid Roman numeral strings into their
1093
+ corresponding integer values while ensuring strict adherence to Roman numeral
1094
+ rules. It handles both standard and subtractive notation (e.g., 'XIV' = 14,
1095
+ 'MCMXCIV' = 1994) and validates input to prevent incorrect sequences.
1096
+
1097
+ Parameters:
1098
+ ──────────────────────────
1099
+ - s (*str*):
1100
+ - A valid Roman numeral string (e.g., 'XIV', 'MCMXCIV').
1101
+
1102
+ - to_str (*bool, optional*):
1103
+ - If `True`, returns the result as a string instead of an integer.
1104
+ - Default is `False`.
1105
+
1106
+ Returns:
1107
+ ──────────────────────────
1108
+ - (*int | str | None*):
1109
+ - The integer representation of the Roman numeral.
1110
+ - If `to_str=True`, returns the value as a string.
1111
+ - Returns `None` if the input is invalid.
1112
+ """
1113
+ num = __parseRomanNumeral(s)
1114
+ return str(num) if num is not None and to_str else num
1115
+
1116
+
1117
+ # ROMAN NUMERALS: CONVERT ROMAN NUMERALS (E.G., "XIV") TO CARDINAL NUMBERS IN WORD FORM (E.G., "FOURTEEN")
1118
+ #───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
1119
+ def romanToWords(s: str):
1120
+ """
1121
+ Converts a Roman numeral string into its spelled-out English words representation.
1122
+
1123
+ This function first converts a Roman numeral into an integer and then transforms
1124
+ it into its full English word equivalent. It ensures strict adherence to Roman
1125
+ numeral rules and returns a readable word-based representation.
1126
+
1127
+ Parameters:
1128
+ ──────────────────────────
1129
+ - s (*str*):
1130
+ - A valid Roman numeral string (e.g., 'XIV', 'MCMXCIV').
1131
+
1132
+ Returns:
1133
+ ──────────────────────────
1134
+ - (*str | None*):
1135
+ - The spelled-out English words for the given Roman numeral.
1136
+ - Returns `None` if the input is invalid.
1137
+ """
1138
+ num = __parseRomanNumeral(s)
1139
+ return intToWords(num) if num is not None else None
1140
+
1141
+
1142
+
1143
+
1144
+ __all__ = [
1145
+ # "replaceNumericValue",
1146
+ "wordsToInt",
1147
+ "ordinalSuffix",
1148
+ "intToWords",
1149
+ "intToOrdinalWords",
1150
+ "stripOrdinalSuffix",
1151
+ "ordinalWordsToInt",
1152
+ "stringToInt",
1153
+ "extractNumericValue",
1154
+ "romanToWords",
1155
+ "romanToInt",
1156
+ "insertSep",
1157
+ "formatDecimal",
1158
+ ]
1159
+
1160
+
1161
+
1162
+
1163
+
1164
+
1165
+
1166
+
1167
+
1168
+
1169
+
1170
+
1171
+
1172
+
1173
+
1174
+
1175
+
1176
+
1177
+
1178
+
1179
+
1180
+
@@ -0,0 +1,110 @@
1
+ Metadata-Version: 2.1
2
+ Name: numbr
3
+ Version: 1.0.0
4
+ Summary: A comprehensive Python library for parsing and converting numbers between numeric, word, and ordinal formats.
5
+ Home-page: https://github.com/cedricmoorejr/numbr/tree/v1.0.0
6
+ Author: Cedric Moore Jr.
7
+ Author-email: cedricmoorejunior5@gmail.com
8
+ License: MIT
9
+ Classifier: Programming Language :: Python :: 3
10
+ Classifier: Operating System :: Microsoft :: Windows
11
+ Classifier: Operating System :: POSIX :: Linux
12
+ Classifier: Operating System :: MacOS :: MacOS X
13
+ Classifier: Development Status :: 5 - Production/Stable
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Natural Language :: English
16
+ Classifier: License :: OSI Approved :: MIT License
17
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
18
+ Classifier: Programming Language :: Python :: 3.8
19
+ Classifier: Programming Language :: Python :: 3.9
20
+ Classifier: Programming Language :: Python :: 3.10
21
+ Requires-Python: >=3.6
22
+ Description-Content-Type: text/markdown
23
+
24
+ # numbr
25
+
26
+ **numbr** is a Python library designed for parsing and converting numbers written in English. It simplifies working with numbers by converting between spelled-out forms, ordinal forms, and numeric representations.
27
+
28
+ ---
29
+
30
+ ## Key Features
31
+
32
+ - Convert spelled-out cardinal numbers into integers (e.g., `"one hundred twenty-three"` → `123`).
33
+ - Convert integers into their spelled-out English words (e.g., `123` → `"one hundred twenty-three"`).
34
+ - Convert ordinal words to numeric ordinals (e.g., `"twenty-first"` → `"21st"` or `21`).
35
+ - Extract numeric values from text strings.
36
+ - Handle negative numbers, hyphenated numbers, and large numbers (up to quintillions).
37
+
38
+ ---
39
+
40
+ ## Installation
41
+
42
+ Install `numbr` using `pip`:
43
+
44
+ ```bash
45
+ pip install numbr
46
+ ```
47
+
48
+ ---
49
+
50
+ ## Usage Examples
51
+
52
+ ### Convert Words to Integer
53
+
54
+ ```python
55
+ import numbr
56
+
57
+ print(numbr.wordsToInt("one thousand two hundred thirty-four"))
58
+ # Output: 1234
59
+ ```
60
+
61
+ ### Convert Integer to Words
62
+
63
+ ```python
64
+ print(numbr.intToWords(5678))
65
+ # Output: "five thousand six hundred seventy-eight"
66
+ ```
67
+
68
+ ### Convert Ordinal Words to Numeric Form
69
+
70
+ ```python
71
+ print(numbr.ordinalWordsToInt("forty-second"))
72
+ # Output: "42nd"
73
+
74
+ print(numbr.ordinalWordsToInt("forty-second", to_num=True))
75
+ # Output: 42
76
+ ```
77
+
78
+ ### Extract Numeric Values from Strings
79
+
80
+ ```python
81
+ print(numbr.extractNumericValue("I have twenty apples and 13 oranges."))
82
+ # Output: 20
83
+ ```
84
+
85
+ ---
86
+
87
+ ## Terminology
88
+
89
+ - **Cardinal Numbers**: Represent quantity (e.g., "one", "twenty-five", "1,234").
90
+ - **Ordinal Numbers**: Represent position or order (e.g., "first", "twenty-first", "3rd").
91
+ - **Ordinal Suffix**: Letters added to numbers indicating position ("st", "nd", "rd", "th").
92
+
93
+ ---
94
+
95
+ ## License
96
+
97
+ This project is licensed under the MIT License. See the [LICENSE](LICENSE) file for details.
98
+
99
+ ---
100
+
101
+ ## Contributing
102
+
103
+ Contributions are welcome! Feel free to open issues or submit pull requests to improve `numbr`.
104
+
105
+ ---
106
+
107
+ ## Contact
108
+
109
+ For questions or feedback, please open an issue on the project's GitHub repository.
110
+
@@ -0,0 +1,6 @@
1
+ numbr/__init__.py,sha256=ih6LAoWE2-ZfrkAWPcUIyYTh2VMVF9uysdHZSh7FVeQ,991
2
+ numbr/engine.py,sha256=pnvuEGaihKJe-5-2O3iuwbSMVGgoOvBPDom1fcvofJ0,53157
3
+ numbr-1.0.0.dist-info/METADATA,sha256=2CwNXybHv-LaK6ELNybV3gprG-QhdW1znknmyTn46XA,3101
4
+ numbr-1.0.0.dist-info/WHEEL,sha256=pkctZYzUS4AYVn6dJ-7367OJZivF2e8RA9b_ZBjif18,92
5
+ numbr-1.0.0.dist-info/top_level.txt,sha256=FoAqdARWQS0e47iqVRzh8HclyF_xyc_0saSAnjbfSco,6
6
+ numbr-1.0.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: bdist_wheel (0.40.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1 @@
1
+ numbr