tablas-python 0.1.4__tar.gz → 0.1.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tablas_python-0.1.4 → tablas_python-0.1.5}/PKG-INFO +1 -1
- {tablas_python-0.1.4 → tablas_python-0.1.5}/pyproject.toml +1 -1
- {tablas_python-0.1.4 → tablas_python-0.1.5}/tablas_python/__init__.py +1 -1
- {tablas_python-0.1.4 → tablas_python-0.1.5}/tablas_python.egg-info/PKG-INFO +1 -1
- {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/data_helpers.py +30 -25
- {tablas_python-0.1.4 → tablas_python-0.1.5}/LICENSE +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/README.md +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/helpers/__init__.py +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/helpers/display_helper.py +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/helpers/table_manager.py +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/setup.cfg +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/tablas_python/cli.py +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/tablas_python.egg-info/SOURCES.txt +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/tablas_python.egg-info/dependency_links.txt +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/tablas_python.egg-info/entry_points.txt +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/tablas_python.egg-info/requires.txt +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/tablas_python.egg-info/top_level.txt +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/__init__.py +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/batch_processor.py +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/excel_extractor.py +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/excel_writer.py +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/exporter.py +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/file_utils.py +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/pdf_extractor.py +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/sqlite_extractor.py +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/table_cleaner.py +0 -0
- {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/validator.py +0 -0
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "tablas-python"
|
|
7
|
-
version = "0.1.
|
|
7
|
+
version = "0.1.5"
|
|
8
8
|
description = "Suite modular para extraer, transformar, conciliar y exportar tablas desde PDF, Excel y SQLite a pandas."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
authors = [
|
|
@@ -14,27 +14,30 @@ import numpy as np
|
|
|
14
14
|
# 1. PARSEO Y LIMPIEZA DE NÚMEROS Y FECHAS
|
|
15
15
|
# ==========================================
|
|
16
16
|
|
|
17
|
-
def limpiar_numero(val: Any) -> Union[float, int, Any]:
|
|
17
|
+
def limpiar_numero(val: Any, default: Any = np.nan) -> Union[float, int, Any]:
|
|
18
18
|
"""
|
|
19
|
-
Convierte cualquier
|
|
20
|
-
Maneja formatos latinoamericanos (1.234,56)
|
|
19
|
+
Convierte cualquier valor numérico, moneda o porcentaje a float o int de forma precisa.
|
|
20
|
+
Maneja formatos latinoamericanos (1.234,56 / 0,056), anglosajones (1,234.56 / 0.056),
|
|
21
21
|
símbolos ($ / € / USD / CLP / %), y negativos entre paréntesis (100) -> -100.
|
|
22
22
|
|
|
23
23
|
Ejemplos:
|
|
24
|
+
limpiar_numero("0,056") -> 0.056
|
|
25
|
+
limpiar_numero("9,999") -> 9.999
|
|
26
|
+
limpiar_numero("89,949") -> 89.949
|
|
27
|
+
limpiar_numero("0,056%") -> 0.056
|
|
24
28
|
limpiar_numero("$ 1.250.000") -> 1250000.0
|
|
25
|
-
limpiar_numero("
|
|
26
|
-
limpiar_numero("(450.50)") -> -450.50
|
|
29
|
+
limpiar_numero("(450,50)") -> -450.50
|
|
27
30
|
"""
|
|
28
31
|
if pd.isna(val) or val is None:
|
|
29
|
-
return
|
|
32
|
+
return default
|
|
30
33
|
if isinstance(val, (int, float, np.number)):
|
|
31
|
-
return val
|
|
34
|
+
return float(val) if isinstance(val, float) else val
|
|
32
35
|
|
|
33
36
|
s = str(val).strip()
|
|
34
37
|
if not s:
|
|
35
|
-
return
|
|
38
|
+
return default
|
|
36
39
|
|
|
37
|
-
# Detectar negativos con paréntesis: (123.45)
|
|
40
|
+
# 1. Detectar negativos con paréntesis: (123.45) o (123,45)
|
|
38
41
|
es_negativo = False
|
|
39
42
|
if s.startswith("(") and s.endswith(")"):
|
|
40
43
|
es_negativo = True
|
|
@@ -43,46 +46,48 @@ def limpiar_numero(val: Any) -> Union[float, int, Any]:
|
|
|
43
46
|
es_negativo = True
|
|
44
47
|
s = s[1:].strip()
|
|
45
48
|
|
|
46
|
-
# Remover símbolos de monedas y
|
|
49
|
+
# 2. Remover símbolos de monedas y texto innecesario
|
|
47
50
|
s = re.sub(r'[\$\€\£\¥\s]|USD|CLP|EUR|UF', '', s, flags=re.IGNORECASE)
|
|
48
51
|
# Remover %
|
|
49
52
|
s = s.replace('%', '').strip()
|
|
50
53
|
|
|
51
54
|
if not s:
|
|
52
|
-
return
|
|
55
|
+
return default
|
|
53
56
|
|
|
54
|
-
# Determinar
|
|
55
|
-
# Caso
|
|
57
|
+
# 3. Determinar separadores decimales y de miles
|
|
58
|
+
# Caso A: Tiene comas y puntos (ej: 1.234.567,89 o 1,234,567.89)
|
|
56
59
|
if '.' in s and ',' in s:
|
|
57
60
|
if s.rfind(',') > s.rfind('.'):
|
|
58
|
-
# Formato latino: 1.234,
|
|
61
|
+
# Formato latino: 1.234.567,89 -> quitar puntos y reemplazar coma por punto
|
|
59
62
|
s = s.replace('.', '').replace(',', '.')
|
|
60
63
|
else:
|
|
61
|
-
# Formato anglo: 1,234.
|
|
64
|
+
# Formato anglo: 1,234,567.89 -> quitar comas
|
|
62
65
|
s = s.replace(',', '')
|
|
66
|
+
|
|
67
|
+
# Caso B: Solo tiene coma(s)
|
|
63
68
|
elif ',' in s:
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
69
|
+
num_comas = s.count(',')
|
|
70
|
+
if num_comas == 1:
|
|
71
|
+
# 1 sola coma: SIEMPRE es separador decimal en español (ej: 0,056 | 9,999 | 89,949 | 123,45)
|
|
67
72
|
s = s.replace(',', '.')
|
|
68
73
|
else:
|
|
69
|
-
#
|
|
74
|
+
# Múltiples comas: separador de miles anglosajón (ej: 1,000,000)
|
|
70
75
|
s = s.replace(',', '')
|
|
76
|
+
|
|
77
|
+
# Caso C: Solo tiene punto(s)
|
|
71
78
|
elif '.' in s:
|
|
72
|
-
|
|
73
|
-
if
|
|
79
|
+
num_puntos = s.count('.')
|
|
80
|
+
if num_puntos > 1:
|
|
81
|
+
# Múltiples puntos: separador de miles latino (ej: 1.000.000)
|
|
74
82
|
s = s.replace('.', '')
|
|
75
|
-
# Si tiene 1 punto y exactamente 3 dígitos después, puede ser miles (ej: 1.500)
|
|
76
|
-
# pero en float estándar suele ser decimal. Lo dejamos a float directo.
|
|
77
83
|
|
|
78
84
|
try:
|
|
79
85
|
num = float(s)
|
|
80
86
|
if es_negativo:
|
|
81
87
|
num = -num
|
|
82
|
-
# Si no tiene decimales reales, retornar float o int
|
|
83
88
|
return num
|
|
84
89
|
except ValueError:
|
|
85
|
-
return
|
|
90
|
+
return default
|
|
86
91
|
|
|
87
92
|
|
|
88
93
|
def limpiar_columnas_numericas(df: pd.DataFrame, columnas: Union[str, List[str]]) -> pd.DataFrame:
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|