tablas-python 0.1.3__tar.gz → 0.1.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tablas_python-0.1.3 → tablas_python-0.1.5}/PKG-INFO +1 -1
- {tablas_python-0.1.3 → tablas_python-0.1.5}/pyproject.toml +1 -1
- {tablas_python-0.1.3 → tablas_python-0.1.5}/tablas_python/__init__.py +1 -1
- {tablas_python-0.1.3 → tablas_python-0.1.5}/tablas_python.egg-info/PKG-INFO +1 -1
- {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/data_helpers.py +83 -36
- {tablas_python-0.1.3 → tablas_python-0.1.5}/LICENSE +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/README.md +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/helpers/__init__.py +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/helpers/display_helper.py +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/helpers/table_manager.py +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/setup.cfg +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/tablas_python/cli.py +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/tablas_python.egg-info/SOURCES.txt +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/tablas_python.egg-info/dependency_links.txt +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/tablas_python.egg-info/entry_points.txt +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/tablas_python.egg-info/requires.txt +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/tablas_python.egg-info/top_level.txt +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/__init__.py +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/batch_processor.py +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/excel_extractor.py +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/excel_writer.py +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/exporter.py +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/file_utils.py +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/pdf_extractor.py +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/sqlite_extractor.py +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/table_cleaner.py +0 -0
- {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/validator.py +0 -0
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "tablas-python"
|
|
7
|
-
version = "0.1.
|
|
7
|
+
version = "0.1.5"
|
|
8
8
|
description = "Suite modular para extraer, transformar, conciliar y exportar tablas desde PDF, Excel y SQLite a pandas."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
authors = [
|
|
@@ -14,27 +14,30 @@ import numpy as np
|
|
|
14
14
|
# 1. PARSEO Y LIMPIEZA DE NÚMEROS Y FECHAS
|
|
15
15
|
# ==========================================
|
|
16
16
|
|
|
17
|
-
def limpiar_numero(val: Any) -> Union[float, int, Any]:
|
|
17
|
+
def limpiar_numero(val: Any, default: Any = np.nan) -> Union[float, int, Any]:
|
|
18
18
|
"""
|
|
19
|
-
Convierte cualquier
|
|
20
|
-
Maneja formatos latinoamericanos (1.234,56)
|
|
19
|
+
Convierte cualquier valor numérico, moneda o porcentaje a float o int de forma precisa.
|
|
20
|
+
Maneja formatos latinoamericanos (1.234,56 / 0,056), anglosajones (1,234.56 / 0.056),
|
|
21
21
|
símbolos ($ / € / USD / CLP / %), y negativos entre paréntesis (100) -> -100.
|
|
22
22
|
|
|
23
23
|
Ejemplos:
|
|
24
|
+
limpiar_numero("0,056") -> 0.056
|
|
25
|
+
limpiar_numero("9,999") -> 9.999
|
|
26
|
+
limpiar_numero("89,949") -> 89.949
|
|
27
|
+
limpiar_numero("0,056%") -> 0.056
|
|
24
28
|
limpiar_numero("$ 1.250.000") -> 1250000.0
|
|
25
|
-
limpiar_numero("
|
|
26
|
-
limpiar_numero("(450.50)") -> -450.50
|
|
29
|
+
limpiar_numero("(450,50)") -> -450.50
|
|
27
30
|
"""
|
|
28
31
|
if pd.isna(val) or val is None:
|
|
29
|
-
return
|
|
32
|
+
return default
|
|
30
33
|
if isinstance(val, (int, float, np.number)):
|
|
31
|
-
return val
|
|
34
|
+
return float(val) if isinstance(val, float) else val
|
|
32
35
|
|
|
33
36
|
s = str(val).strip()
|
|
34
37
|
if not s:
|
|
35
|
-
return
|
|
38
|
+
return default
|
|
36
39
|
|
|
37
|
-
# Detectar negativos con paréntesis: (123.45)
|
|
40
|
+
# 1. Detectar negativos con paréntesis: (123.45) o (123,45)
|
|
38
41
|
es_negativo = False
|
|
39
42
|
if s.startswith("(") and s.endswith(")"):
|
|
40
43
|
es_negativo = True
|
|
@@ -43,46 +46,48 @@ def limpiar_numero(val: Any) -> Union[float, int, Any]:
|
|
|
43
46
|
es_negativo = True
|
|
44
47
|
s = s[1:].strip()
|
|
45
48
|
|
|
46
|
-
# Remover símbolos de monedas y
|
|
49
|
+
# 2. Remover símbolos de monedas y texto innecesario
|
|
47
50
|
s = re.sub(r'[\$\€\£\¥\s]|USD|CLP|EUR|UF', '', s, flags=re.IGNORECASE)
|
|
48
51
|
# Remover %
|
|
49
52
|
s = s.replace('%', '').strip()
|
|
50
53
|
|
|
51
54
|
if not s:
|
|
52
|
-
return
|
|
55
|
+
return default
|
|
53
56
|
|
|
54
|
-
# Determinar
|
|
55
|
-
# Caso
|
|
57
|
+
# 3. Determinar separadores decimales y de miles
|
|
58
|
+
# Caso A: Tiene comas y puntos (ej: 1.234.567,89 o 1,234,567.89)
|
|
56
59
|
if '.' in s and ',' in s:
|
|
57
60
|
if s.rfind(',') > s.rfind('.'):
|
|
58
|
-
# Formato latino: 1.234,
|
|
61
|
+
# Formato latino: 1.234.567,89 -> quitar puntos y reemplazar coma por punto
|
|
59
62
|
s = s.replace('.', '').replace(',', '.')
|
|
60
63
|
else:
|
|
61
|
-
# Formato anglo: 1,234.
|
|
64
|
+
# Formato anglo: 1,234,567.89 -> quitar comas
|
|
62
65
|
s = s.replace(',', '')
|
|
66
|
+
|
|
67
|
+
# Caso B: Solo tiene coma(s)
|
|
63
68
|
elif ',' in s:
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
69
|
+
num_comas = s.count(',')
|
|
70
|
+
if num_comas == 1:
|
|
71
|
+
# 1 sola coma: SIEMPRE es separador decimal en español (ej: 0,056 | 9,999 | 89,949 | 123,45)
|
|
67
72
|
s = s.replace(',', '.')
|
|
68
73
|
else:
|
|
69
|
-
#
|
|
74
|
+
# Múltiples comas: separador de miles anglosajón (ej: 1,000,000)
|
|
70
75
|
s = s.replace(',', '')
|
|
76
|
+
|
|
77
|
+
# Caso C: Solo tiene punto(s)
|
|
71
78
|
elif '.' in s:
|
|
72
|
-
|
|
73
|
-
if
|
|
79
|
+
num_puntos = s.count('.')
|
|
80
|
+
if num_puntos > 1:
|
|
81
|
+
# Múltiples puntos: separador de miles latino (ej: 1.000.000)
|
|
74
82
|
s = s.replace('.', '')
|
|
75
|
-
# Si tiene 1 punto y exactamente 3 dígitos después, puede ser miles (ej: 1.500)
|
|
76
|
-
# pero en float estándar suele ser decimal. Lo dejamos a float directo.
|
|
77
83
|
|
|
78
84
|
try:
|
|
79
85
|
num = float(s)
|
|
80
86
|
if es_negativo:
|
|
81
87
|
num = -num
|
|
82
|
-
# Si no tiene decimales reales, retornar float o int
|
|
83
88
|
return num
|
|
84
89
|
except ValueError:
|
|
85
|
-
return
|
|
90
|
+
return default
|
|
86
91
|
|
|
87
92
|
|
|
88
93
|
def limpiar_columnas_numericas(df: pd.DataFrame, columnas: Union[str, List[str]]) -> pd.DataFrame:
|
|
@@ -141,22 +146,56 @@ def formato_moneda(val: Any, simbolo: str = "$", decimales: int = 0, separador_m
|
|
|
141
146
|
return f"{simbolo} {texto}" if simbolo else texto
|
|
142
147
|
|
|
143
148
|
|
|
144
|
-
def formato_porcentaje(
|
|
149
|
+
def formato_porcentaje(
|
|
150
|
+
val: Any,
|
|
151
|
+
decimales: Optional[int] = None,
|
|
152
|
+
multiplicar_por_100: bool = False,
|
|
153
|
+
separador_decimal: str = ","
|
|
154
|
+
) -> str:
|
|
145
155
|
"""
|
|
146
|
-
Formatea un valor a porcentaje.
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
156
|
+
Formatea un valor a porcentaje de manera exacta.
|
|
157
|
+
|
|
158
|
+
A diferencia de versiones anteriores, NO multiplica arbitrariamente por 100
|
|
159
|
+
a menos que se indique explícitamente con `multiplicar_por_100=True`.
|
|
160
|
+
Por lo tanto, 0.056 se muestra como 0,056% y 5.56 se muestra como 5,56%.
|
|
161
|
+
|
|
162
|
+
Ejemplos:
|
|
163
|
+
formato_porcentaje(5.56) -> "5,56%"
|
|
164
|
+
formato_porcentaje(0.056) -> "0,056%"
|
|
165
|
+
formato_porcentaje("0.056%") -> "0,056%"
|
|
166
|
+
formato_porcentaje(0.056, multiplicar_por_100=True) -> "5,6%"
|
|
167
|
+
formato_porcentaje(15.42, decimales=1) -> "15,4%"
|
|
150
168
|
"""
|
|
169
|
+
if pd.isna(val) or val is None or str(val).strip() == "":
|
|
170
|
+
return ""
|
|
171
|
+
|
|
172
|
+
s_orig = str(val).strip()
|
|
173
|
+
tenia_simbolo_pct = "%" in s_orig
|
|
174
|
+
|
|
151
175
|
num = limpiar_numero(val)
|
|
152
176
|
if pd.isna(num) or not isinstance(num, (int, float)):
|
|
153
|
-
return
|
|
154
|
-
|
|
155
|
-
#
|
|
156
|
-
if
|
|
177
|
+
return s_orig
|
|
178
|
+
|
|
179
|
+
# Solo multiplicar por 100 si el usuario lo pide explícitamente y el valor NO traía ya el '%'
|
|
180
|
+
if multiplicar_por_100 and not tenia_simbolo_pct:
|
|
157
181
|
num = num * 100
|
|
158
182
|
|
|
159
|
-
|
|
183
|
+
if decimales is not None:
|
|
184
|
+
fmt_str = f"{{:.{decimales}f}}"
|
|
185
|
+
txt_num = fmt_str.format(num)
|
|
186
|
+
else:
|
|
187
|
+
# Conservar los decimales significativos del número sin truncar
|
|
188
|
+
if isinstance(num, int) or (isinstance(num, float) and num.is_integer()):
|
|
189
|
+
txt_num = str(int(num))
|
|
190
|
+
else:
|
|
191
|
+
txt_num = f"{num:.6f}".rstrip('0').rstrip('.')
|
|
192
|
+
|
|
193
|
+
if separador_decimal == ",":
|
|
194
|
+
txt_num = txt_num.replace(".", ",")
|
|
195
|
+
else:
|
|
196
|
+
txt_num = txt_num.replace(",", ".")
|
|
197
|
+
|
|
198
|
+
return f"{txt_num}%"
|
|
160
199
|
|
|
161
200
|
|
|
162
201
|
def formato_miles(val: Any, decimales: int = 0) -> str:
|
|
@@ -176,6 +215,8 @@ def formatear_dataframe(
|
|
|
176
215
|
'Ventas': 'moneda',
|
|
177
216
|
'Precio': 'moneda_2dec',
|
|
178
217
|
'Margen': 'porcentaje',
|
|
218
|
+
'Participacion': 'porcentaje_1dec',
|
|
219
|
+
'Ratio': 'ratio_pct',
|
|
179
220
|
'Cantidad': 'miles'
|
|
180
221
|
})
|
|
181
222
|
"""
|
|
@@ -188,7 +229,13 @@ def formatear_dataframe(
|
|
|
188
229
|
elif tipo == 'moneda_2dec':
|
|
189
230
|
df_out[col] = df_out[col].apply(lambda x: formato_moneda(x, decimales=2))
|
|
190
231
|
elif tipo == 'porcentaje':
|
|
191
|
-
df_out[col] = df_out[col].apply(formato_porcentaje)
|
|
232
|
+
df_out[col] = df_out[col].apply(lambda x: formato_porcentaje(x))
|
|
233
|
+
elif tipo == 'porcentaje_1dec':
|
|
234
|
+
df_out[col] = df_out[col].apply(lambda x: formato_porcentaje(x, decimales=1))
|
|
235
|
+
elif tipo == 'porcentaje_2dec':
|
|
236
|
+
df_out[col] = df_out[col].apply(lambda x: formato_porcentaje(x, decimales=2))
|
|
237
|
+
elif tipo == 'ratio_pct':
|
|
238
|
+
df_out[col] = df_out[col].apply(lambda x: formato_porcentaje(x, multiplicar_por_100=True, decimales=1))
|
|
192
239
|
elif tipo == 'miles':
|
|
193
240
|
df_out[col] = df_out[col].apply(formato_miles)
|
|
194
241
|
return df_out
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|