tablas-python 0.1.3__tar.gz → 0.1.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {tablas_python-0.1.3 → tablas_python-0.1.5}/PKG-INFO +1 -1
  2. {tablas_python-0.1.3 → tablas_python-0.1.5}/pyproject.toml +1 -1
  3. {tablas_python-0.1.3 → tablas_python-0.1.5}/tablas_python/__init__.py +1 -1
  4. {tablas_python-0.1.3 → tablas_python-0.1.5}/tablas_python.egg-info/PKG-INFO +1 -1
  5. {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/data_helpers.py +83 -36
  6. {tablas_python-0.1.3 → tablas_python-0.1.5}/LICENSE +0 -0
  7. {tablas_python-0.1.3 → tablas_python-0.1.5}/README.md +0 -0
  8. {tablas_python-0.1.3 → tablas_python-0.1.5}/helpers/__init__.py +0 -0
  9. {tablas_python-0.1.3 → tablas_python-0.1.5}/helpers/display_helper.py +0 -0
  10. {tablas_python-0.1.3 → tablas_python-0.1.5}/helpers/table_manager.py +0 -0
  11. {tablas_python-0.1.3 → tablas_python-0.1.5}/setup.cfg +0 -0
  12. {tablas_python-0.1.3 → tablas_python-0.1.5}/tablas_python/cli.py +0 -0
  13. {tablas_python-0.1.3 → tablas_python-0.1.5}/tablas_python.egg-info/SOURCES.txt +0 -0
  14. {tablas_python-0.1.3 → tablas_python-0.1.5}/tablas_python.egg-info/dependency_links.txt +0 -0
  15. {tablas_python-0.1.3 → tablas_python-0.1.5}/tablas_python.egg-info/entry_points.txt +0 -0
  16. {tablas_python-0.1.3 → tablas_python-0.1.5}/tablas_python.egg-info/requires.txt +0 -0
  17. {tablas_python-0.1.3 → tablas_python-0.1.5}/tablas_python.egg-info/top_level.txt +0 -0
  18. {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/__init__.py +0 -0
  19. {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/batch_processor.py +0 -0
  20. {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/excel_extractor.py +0 -0
  21. {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/excel_writer.py +0 -0
  22. {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/exporter.py +0 -0
  23. {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/file_utils.py +0 -0
  24. {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/pdf_extractor.py +0 -0
  25. {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/sqlite_extractor.py +0 -0
  26. {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/table_cleaner.py +0 -0
  27. {tablas_python-0.1.3 → tablas_python-0.1.5}/utils/validator.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tablas-python
3
- Version: 0.1.3
3
+ Version: 0.1.5
4
4
  Summary: Suite modular para extraer, transformar, conciliar y exportar tablas desde PDF, Excel y SQLite a pandas.
5
5
  Author: Reciba
6
6
  License: MIT
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "tablas-python"
7
- version = "0.1.3"
7
+ version = "0.1.5"
8
8
  description = "Suite modular para extraer, transformar, conciliar y exportar tablas desde PDF, Excel y SQLite a pandas."
9
9
  readme = "README.md"
10
10
  authors = [
@@ -32,7 +32,7 @@ from utils.data_helpers import (
32
32
  conciliar_tablas,
33
33
  )
34
34
 
35
- __version__ = "0.1.3"
35
+ __version__ = "0.1.5"
36
36
 
37
37
  __all__ = [
38
38
  "TableManager",
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tablas-python
3
- Version: 0.1.3
3
+ Version: 0.1.5
4
4
  Summary: Suite modular para extraer, transformar, conciliar y exportar tablas desde PDF, Excel y SQLite a pandas.
5
5
  Author: Reciba
6
6
  License: MIT
@@ -14,27 +14,30 @@ import numpy as np
14
14
  # 1. PARSEO Y LIMPIEZA DE NÚMEROS Y FECHAS
15
15
  # ==========================================
16
16
 
17
- def limpiar_numero(val: Any) -> Union[float, int, Any]:
17
+ def limpiar_numero(val: Any, default: Any = np.nan) -> Union[float, int, Any]:
18
18
  """
19
- Convierte cualquier cadena con formato numérico, moneda o porcentaje a float o int.
20
- Maneja formatos latinoamericanos (1.234,56) y anglosajones (1,234.56),
19
+ Convierte cualquier valor numérico, moneda o porcentaje a float o int de forma precisa.
20
+ Maneja formatos latinoamericanos (1.234,56 / 0,056), anglosajones (1,234.56 / 0.056),
21
21
  símbolos ($ / € / USD / CLP / %), y negativos entre paréntesis (100) -> -100.
22
22
 
23
23
  Ejemplos:
24
+ limpiar_numero("0,056") -> 0.056
25
+ limpiar_numero("9,999") -> 9.999
26
+ limpiar_numero("89,949") -> 89.949
27
+ limpiar_numero("0,056%") -> 0.056
24
28
  limpiar_numero("$ 1.250.000") -> 1250000.0
25
- limpiar_numero("18,5 %") -> 18.5
26
- limpiar_numero("(450.50)") -> -450.50
29
+ limpiar_numero("(450,50)") -> -450.50
27
30
  """
28
31
  if pd.isna(val) or val is None:
29
- return np.nan
32
+ return default
30
33
  if isinstance(val, (int, float, np.number)):
31
- return val
34
+ return float(val) if isinstance(val, float) else val
32
35
 
33
36
  s = str(val).strip()
34
37
  if not s:
35
- return np.nan
38
+ return default
36
39
 
37
- # Detectar negativos con paréntesis: (123.45)
40
+ # 1. Detectar negativos con paréntesis: (123.45) o (123,45)
38
41
  es_negativo = False
39
42
  if s.startswith("(") and s.endswith(")"):
40
43
  es_negativo = True
@@ -43,46 +46,48 @@ def limpiar_numero(val: Any) -> Union[float, int, Any]:
43
46
  es_negativo = True
44
47
  s = s[1:].strip()
45
48
 
46
- # Remover símbolos de monedas y palabras comunes
49
+ # 2. Remover símbolos de monedas y texto innecesario
47
50
  s = re.sub(r'[\$\€\£\¥\s]|USD|CLP|EUR|UF', '', s, flags=re.IGNORECASE)
48
51
  # Remover %
49
52
  s = s.replace('%', '').strip()
50
53
 
51
54
  if not s:
52
- return np.nan
55
+ return default
53
56
 
54
- # Determinar si el separador decimal es coma o punto
55
- # Caso 1: Tiene comas y puntos (ej: 1.234.567,89 o 1,234,567.89)
57
+ # 3. Determinar separadores decimales y de miles
58
+ # Caso A: Tiene comas y puntos (ej: 1.234.567,89 o 1,234,567.89)
56
59
  if '.' in s and ',' in s:
57
60
  if s.rfind(',') > s.rfind('.'):
58
- # Formato latino: 1.234,56 -> quitar puntos y coma a punto
61
+ # Formato latino: 1.234.567,89 -> quitar puntos y reemplazar coma por punto
59
62
  s = s.replace('.', '').replace(',', '.')
60
63
  else:
61
- # Formato anglo: 1,234.56 -> quitar comas
64
+ # Formato anglo: 1,234,567.89 -> quitar comas
62
65
  s = s.replace(',', '')
66
+
67
+ # Caso B: Solo tiene coma(s)
63
68
  elif ',' in s:
64
- # Solo tiene comas. Si tiene 1 coma y max 2 decimales al final: 1234,56
65
- partes = s.split(',')
66
- if len(partes) == 2 and len(partes[1]) <= 2:
69
+ num_comas = s.count(',')
70
+ if num_comas == 1:
71
+ # 1 sola coma: SIEMPRE es separador decimal en español (ej: 0,056 | 9,999 | 89,949 | 123,45)
67
72
  s = s.replace(',', '.')
68
73
  else:
69
- # Es separador de miles: 1,000,000
74
+ # Múltiples comas: separador de miles anglosajón (ej: 1,000,000)
70
75
  s = s.replace(',', '')
76
+
77
+ # Caso C: Solo tiene punto(s)
71
78
  elif '.' in s:
72
- # Solo tiene puntos. Si tiene múltiples puntos: 1.000.000
73
- if s.count('.') > 1:
79
+ num_puntos = s.count('.')
80
+ if num_puntos > 1:
81
+ # Múltiples puntos: separador de miles latino (ej: 1.000.000)
74
82
  s = s.replace('.', '')
75
- # Si tiene 1 punto y exactamente 3 dígitos después, puede ser miles (ej: 1.500)
76
- # pero en float estándar suele ser decimal. Lo dejamos a float directo.
77
83
 
78
84
  try:
79
85
  num = float(s)
80
86
  if es_negativo:
81
87
  num = -num
82
- # Si no tiene decimales reales, retornar float o int
83
88
  return num
84
89
  except ValueError:
85
- return val
90
+ return default
86
91
 
87
92
 
88
93
  def limpiar_columnas_numericas(df: pd.DataFrame, columnas: Union[str, List[str]]) -> pd.DataFrame:
@@ -141,22 +146,56 @@ def formato_moneda(val: Any, simbolo: str = "$", decimales: int = 0, separador_m
141
146
  return f"{simbolo} {texto}" if simbolo else texto
142
147
 
143
148
 
144
- def formato_porcentaje(val: Any, decimales: int = 1) -> str:
149
+ def formato_porcentaje(
150
+ val: Any,
151
+ decimales: Optional[int] = None,
152
+ multiplicar_por_100: bool = False,
153
+ separador_decimal: str = ","
154
+ ) -> str:
145
155
  """
146
- Formatea un valor a porcentaje.
147
- Ejemplo:
148
- formato_porcentaje(15.42) -> "15.4%"
149
- formato_porcentaje(0.1542, es_ratio=True)
156
+ Formatea un valor a porcentaje de manera exacta.
157
+
158
+ A diferencia de versiones anteriores, NO multiplica arbitrariamente por 100
159
+ a menos que se indique explícitamente con `multiplicar_por_100=True`.
160
+ Por lo tanto, 0.056 se muestra como 0,056% y 5.56 se muestra como 5,56%.
161
+
162
+ Ejemplos:
163
+ formato_porcentaje(5.56) -> "5,56%"
164
+ formato_porcentaje(0.056) -> "0,056%"
165
+ formato_porcentaje("0.056%") -> "0,056%"
166
+ formato_porcentaje(0.056, multiplicar_por_100=True) -> "5,6%"
167
+ formato_porcentaje(15.42, decimales=1) -> "15,4%"
150
168
  """
169
+ if pd.isna(val) or val is None or str(val).strip() == "":
170
+ return ""
171
+
172
+ s_orig = str(val).strip()
173
+ tenia_simbolo_pct = "%" in s_orig
174
+
151
175
  num = limpiar_numero(val)
152
176
  if pd.isna(num) or not isinstance(num, (int, float)):
153
- return "" if pd.isna(val) else str(val)
154
-
155
- # Si el valor está entre 0 y 1 (ratio), convertir a base 100
156
- if -1.0 <= num <= 1.0 and num != 0:
177
+ return s_orig
178
+
179
+ # Solo multiplicar por 100 si el usuario lo pide explícitamente y el valor NO traía ya el '%'
180
+ if multiplicar_por_100 and not tenia_simbolo_pct:
157
181
  num = num * 100
158
182
 
159
- return f"{num:.{decimales}f}%"
183
+ if decimales is not None:
184
+ fmt_str = f"{{:.{decimales}f}}"
185
+ txt_num = fmt_str.format(num)
186
+ else:
187
+ # Conservar los decimales significativos del número sin truncar
188
+ if isinstance(num, int) or (isinstance(num, float) and num.is_integer()):
189
+ txt_num = str(int(num))
190
+ else:
191
+ txt_num = f"{num:.6f}".rstrip('0').rstrip('.')
192
+
193
+ if separador_decimal == ",":
194
+ txt_num = txt_num.replace(".", ",")
195
+ else:
196
+ txt_num = txt_num.replace(",", ".")
197
+
198
+ return f"{txt_num}%"
160
199
 
161
200
 
162
201
  def formato_miles(val: Any, decimales: int = 0) -> str:
@@ -176,6 +215,8 @@ def formatear_dataframe(
176
215
  'Ventas': 'moneda',
177
216
  'Precio': 'moneda_2dec',
178
217
  'Margen': 'porcentaje',
218
+ 'Participacion': 'porcentaje_1dec',
219
+ 'Ratio': 'ratio_pct',
179
220
  'Cantidad': 'miles'
180
221
  })
181
222
  """
@@ -188,7 +229,13 @@ def formatear_dataframe(
188
229
  elif tipo == 'moneda_2dec':
189
230
  df_out[col] = df_out[col].apply(lambda x: formato_moneda(x, decimales=2))
190
231
  elif tipo == 'porcentaje':
191
- df_out[col] = df_out[col].apply(formato_porcentaje)
232
+ df_out[col] = df_out[col].apply(lambda x: formato_porcentaje(x))
233
+ elif tipo == 'porcentaje_1dec':
234
+ df_out[col] = df_out[col].apply(lambda x: formato_porcentaje(x, decimales=1))
235
+ elif tipo == 'porcentaje_2dec':
236
+ df_out[col] = df_out[col].apply(lambda x: formato_porcentaje(x, decimales=2))
237
+ elif tipo == 'ratio_pct':
238
+ df_out[col] = df_out[col].apply(lambda x: formato_porcentaje(x, multiplicar_por_100=True, decimales=1))
192
239
  elif tipo == 'miles':
193
240
  df_out[col] = df_out[col].apply(formato_miles)
194
241
  return df_out
File without changes
File without changes
File without changes