tablas-python 0.1.6__tar.gz → 0.1.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tablas_python-0.1.6 → tablas_python-0.1.7}/PKG-INFO +1 -1
- {tablas_python-0.1.6 → tablas_python-0.1.7}/pyproject.toml +1 -1
- {tablas_python-0.1.6 → tablas_python-0.1.7}/tablas_python/__init__.py +1 -1
- {tablas_python-0.1.6 → tablas_python-0.1.7}/tablas_python.egg-info/PKG-INFO +1 -1
- {tablas_python-0.1.6 → tablas_python-0.1.7}/utils/data_helpers.py +24 -9
- {tablas_python-0.1.6 → tablas_python-0.1.7}/LICENSE +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/README.md +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/helpers/__init__.py +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/helpers/display_helper.py +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/helpers/table_manager.py +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/setup.cfg +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/tablas_python/cli.py +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/tablas_python.egg-info/SOURCES.txt +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/tablas_python.egg-info/dependency_links.txt +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/tablas_python.egg-info/entry_points.txt +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/tablas_python.egg-info/requires.txt +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/tablas_python.egg-info/top_level.txt +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/utils/__init__.py +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/utils/batch_processor.py +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/utils/excel_extractor.py +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/utils/excel_writer.py +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/utils/exporter.py +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/utils/file_utils.py +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/utils/pdf_extractor.py +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/utils/sqlite_extractor.py +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/utils/table_cleaner.py +0 -0
- {tablas_python-0.1.6 → tablas_python-0.1.7}/utils/validator.py +0 -0
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "tablas-python"
|
|
7
|
-
version = "0.1.
|
|
7
|
+
version = "0.1.7"
|
|
8
8
|
description = "Suite modular para extraer, transformar, conciliar y exportar tablas desde PDF, Excel y SQLite a pandas."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
authors = [
|
|
@@ -14,17 +14,20 @@ import numpy as np
|
|
|
14
14
|
# 1. PARSEO Y LIMPIEZA DE NÚMEROS Y FECHAS
|
|
15
15
|
# ==========================================
|
|
16
16
|
|
|
17
|
-
def limpiar_numero(val: Any, default: Any =
|
|
17
|
+
def limpiar_numero(val: Any, default: Any = 0) -> Union[float, int, Any]:
|
|
18
18
|
"""
|
|
19
19
|
Convierte cualquier valor numérico, moneda o porcentaje a float o int de forma precisa.
|
|
20
|
+
Si el valor está vacío, es nulo o texto sin número (ej: '-', 'N/A', 'null'), retorna 0 (o el default indicado).
|
|
21
|
+
|
|
20
22
|
Maneja formatos latinoamericanos (1.234,56 / 0,056), anglosajones (1,234.56 / 0.056),
|
|
21
23
|
símbolos ($ / € / USD / CLP / %), y negativos entre paréntesis (100) -> -100.
|
|
22
24
|
|
|
23
25
|
Ejemplos:
|
|
24
26
|
limpiar_numero("0,056") -> 0.056
|
|
25
|
-
limpiar_numero("
|
|
26
|
-
limpiar_numero("
|
|
27
|
-
limpiar_numero("
|
|
27
|
+
limpiar_numero("") -> 0
|
|
28
|
+
limpiar_numero("-") -> 0
|
|
29
|
+
limpiar_numero("N/A") -> 0
|
|
30
|
+
limpiar_numero(None) -> 0
|
|
28
31
|
limpiar_numero("$ 1.250.000") -> 1250000.0
|
|
29
32
|
limpiar_numero("(450,50)") -> -450.50
|
|
30
33
|
"""
|
|
@@ -34,7 +37,7 @@ def limpiar_numero(val: Any, default: Any = np.nan) -> Union[float, int, Any]:
|
|
|
34
37
|
return float(val) if isinstance(val, float) else val
|
|
35
38
|
|
|
36
39
|
s = str(val).strip()
|
|
37
|
-
if not s:
|
|
40
|
+
if not s or s.lower() in ['', '-', '--', 'n/a', 'na', 'null', 'none', 's/i', 's/n']:
|
|
38
41
|
return default
|
|
39
42
|
|
|
40
43
|
# 1. Detectar negativos con paréntesis: (123.45) o (123,45)
|
|
@@ -80,6 +83,13 @@ def limpiar_numero(val: Any, default: Any = np.nan) -> Union[float, int, Any]:
|
|
|
80
83
|
if num_puntos > 1:
|
|
81
84
|
# Múltiples puntos: separador de miles latino (ej: 1.000.000)
|
|
82
85
|
s = s.replace('.', '')
|
|
86
|
+
else:
|
|
87
|
+
# 1 solo punto (ej: 250.000, 39.990 vs 0.056, 12.5)
|
|
88
|
+
partes = s.split('.')
|
|
89
|
+
if len(partes) == 2:
|
|
90
|
+
# Si no empieza con '0' y tiene exactamente 3 dígitos tras el punto, es separador de miles (ej: 250.000)
|
|
91
|
+
if partes[0] not in ['0', '-0'] and len(partes[1]) == 3:
|
|
92
|
+
s = s.replace('.', '')
|
|
83
93
|
|
|
84
94
|
try:
|
|
85
95
|
num = float(s)
|
|
@@ -90,17 +100,22 @@ def limpiar_numero(val: Any, default: Any = np.nan) -> Union[float, int, Any]:
|
|
|
90
100
|
return default
|
|
91
101
|
|
|
92
102
|
|
|
93
|
-
def limpiar_columnas_numericas(
|
|
103
|
+
def limpiar_columnas_numericas(
|
|
104
|
+
df: pd.DataFrame,
|
|
105
|
+
columnas: Union[str, List[str]],
|
|
106
|
+
rellenar_nulos: Any = 0
|
|
107
|
+
) -> pd.DataFrame:
|
|
94
108
|
"""
|
|
95
|
-
Convierte una o más columnas a tipo numérico (float) limpiando caracteres extra
|
|
109
|
+
Convierte una o más columnas a tipo numérico (float) limpiando caracteres extra
|
|
110
|
+
y rellenando valores vacíos o nulos con 0 (o el valor indicado).
|
|
96
111
|
"""
|
|
97
112
|
df_out = df.copy()
|
|
98
113
|
if isinstance(columnas, str):
|
|
99
114
|
columnas = [columnas]
|
|
100
115
|
for col in columnas:
|
|
101
116
|
if col in df_out.columns:
|
|
102
|
-
df_out[col] = df_out[col].apply(limpiar_numero)
|
|
103
|
-
df_out[col] = pd.to_numeric(df_out[col], errors='coerce')
|
|
117
|
+
df_out[col] = df_out[col].apply(lambda x: limpiar_numero(x, default=rellenar_nulos))
|
|
118
|
+
df_out[col] = pd.to_numeric(df_out[col], errors='coerce').fillna(rellenar_nulos)
|
|
104
119
|
return df_out
|
|
105
120
|
|
|
106
121
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|