tablas-python 0.1.4__tar.gz → 0.1.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {tablas_python-0.1.4 → tablas_python-0.1.6}/PKG-INFO +1 -1
  2. {tablas_python-0.1.4 → tablas_python-0.1.6}/helpers/__init__.py +4 -1
  3. {tablas_python-0.1.4 → tablas_python-0.1.6}/pyproject.toml +1 -1
  4. {tablas_python-0.1.4 → tablas_python-0.1.6}/tablas_python/__init__.py +3 -1
  5. {tablas_python-0.1.4 → tablas_python-0.1.6}/tablas_python.egg-info/PKG-INFO +1 -1
  6. {tablas_python-0.1.4 → tablas_python-0.1.6}/utils/__init__.py +2 -0
  7. {tablas_python-0.1.4 → tablas_python-0.1.6}/utils/data_helpers.py +48 -27
  8. {tablas_python-0.1.4 → tablas_python-0.1.6}/LICENSE +0 -0
  9. {tablas_python-0.1.4 → tablas_python-0.1.6}/README.md +0 -0
  10. {tablas_python-0.1.4 → tablas_python-0.1.6}/helpers/display_helper.py +0 -0
  11. {tablas_python-0.1.4 → tablas_python-0.1.6}/helpers/table_manager.py +0 -0
  12. {tablas_python-0.1.4 → tablas_python-0.1.6}/setup.cfg +0 -0
  13. {tablas_python-0.1.4 → tablas_python-0.1.6}/tablas_python/cli.py +0 -0
  14. {tablas_python-0.1.4 → tablas_python-0.1.6}/tablas_python.egg-info/SOURCES.txt +0 -0
  15. {tablas_python-0.1.4 → tablas_python-0.1.6}/tablas_python.egg-info/dependency_links.txt +0 -0
  16. {tablas_python-0.1.4 → tablas_python-0.1.6}/tablas_python.egg-info/entry_points.txt +0 -0
  17. {tablas_python-0.1.4 → tablas_python-0.1.6}/tablas_python.egg-info/requires.txt +0 -0
  18. {tablas_python-0.1.4 → tablas_python-0.1.6}/tablas_python.egg-info/top_level.txt +0 -0
  19. {tablas_python-0.1.4 → tablas_python-0.1.6}/utils/batch_processor.py +0 -0
  20. {tablas_python-0.1.4 → tablas_python-0.1.6}/utils/excel_extractor.py +0 -0
  21. {tablas_python-0.1.4 → tablas_python-0.1.6}/utils/excel_writer.py +0 -0
  22. {tablas_python-0.1.4 → tablas_python-0.1.6}/utils/exporter.py +0 -0
  23. {tablas_python-0.1.4 → tablas_python-0.1.6}/utils/file_utils.py +0 -0
  24. {tablas_python-0.1.4 → tablas_python-0.1.6}/utils/pdf_extractor.py +0 -0
  25. {tablas_python-0.1.4 → tablas_python-0.1.6}/utils/sqlite_extractor.py +0 -0
  26. {tablas_python-0.1.4 → tablas_python-0.1.6}/utils/table_cleaner.py +0 -0
  27. {tablas_python-0.1.4 → tablas_python-0.1.6}/utils/validator.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tablas-python
3
- Version: 0.1.4
3
+ Version: 0.1.6
4
4
  Summary: Suite modular para extraer, transformar, conciliar y exportar tablas desde PDF, Excel y SQLite a pandas.
5
5
  Author: Reciba
6
6
  License: MIT
@@ -7,7 +7,7 @@ from .display_helper import DisplayHelper, mostrar_tabla
7
7
  from utils.batch_processor import unir_archivos_carpeta
8
8
  from utils.excel_writer import escribir_en_excel
9
9
  from utils.validator import validar_dataframe, reporte_calidad, detectar_duplicados
10
- from utils.data_helpers import buscar_v, conciliar_tablas, obtener_celda, modificar_celda
10
+ from utils.data_helpers import buscar_v, conciliar_tablas, obtener_celda, modificar_celda, formato_clp, formato_porcentaje, formato_moneda
11
11
 
12
12
  __all__ = [
13
13
  "TableManager",
@@ -23,6 +23,9 @@ __all__ = [
23
23
  "conciliar_tablas",
24
24
  "obtener_celda",
25
25
  "modificar_celda",
26
+ "formato_clp",
27
+ "formato_porcentaje",
28
+ "formato_moneda",
26
29
  "DisplayHelper",
27
30
  "mostrar_tabla",
28
31
  ]
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "tablas-python"
7
- version = "0.1.4"
7
+ version = "0.1.6"
8
8
  description = "Suite modular para extraer, transformar, conciliar y exportar tablas desde PDF, Excel y SQLite a pandas."
9
9
  readme = "README.md"
10
10
  authors = [
@@ -18,6 +18,7 @@ from utils.data_helpers import (
18
18
  limpiar_columnas_numericas,
19
19
  normalizar_fechas,
20
20
  formato_moneda,
21
+ formato_clp,
21
22
  formato_porcentaje,
22
23
  formato_miles,
23
24
  formatear_dataframe,
@@ -32,7 +33,7 @@ from utils.data_helpers import (
32
33
  conciliar_tablas,
33
34
  )
34
35
 
35
- __version__ = "0.1.4"
36
+ __version__ = "0.1.6"
36
37
 
37
38
  __all__ = [
38
39
  "TableManager",
@@ -54,6 +55,7 @@ __all__ = [
54
55
  "limpiar_columnas_numericas",
55
56
  "normalizar_fechas",
56
57
  "formato_moneda",
58
+ "formato_clp",
57
59
  "formato_porcentaje",
58
60
  "formato_miles",
59
61
  "formatear_dataframe",
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tablas-python
3
- Version: 0.1.4
3
+ Version: 0.1.6
4
4
  Summary: Suite modular para extraer, transformar, conciliar y exportar tablas desde PDF, Excel y SQLite a pandas.
5
5
  Author: Reciba
6
6
  License: MIT
@@ -17,6 +17,7 @@ from .data_helpers import (
17
17
  limpiar_columnas_numericas,
18
18
  normalizar_fechas,
19
19
  formato_moneda,
20
+ formato_clp,
20
21
  formato_porcentaje,
21
22
  formato_miles,
22
23
  formatear_dataframe,
@@ -56,6 +57,7 @@ __all__ = [
56
57
  "limpiar_columnas_numericas",
57
58
  "normalizar_fechas",
58
59
  "formato_moneda",
60
+ "formato_clp",
59
61
  "formato_porcentaje",
60
62
  "formato_miles",
61
63
  "formatear_dataframe",
@@ -14,27 +14,30 @@ import numpy as np
14
14
  # 1. PARSEO Y LIMPIEZA DE NÚMEROS Y FECHAS
15
15
  # ==========================================
16
16
 
17
- def limpiar_numero(val: Any) -> Union[float, int, Any]:
17
+ def limpiar_numero(val: Any, default: Any = np.nan) -> Union[float, int, Any]:
18
18
  """
19
- Convierte cualquier cadena con formato numérico, moneda o porcentaje a float o int.
20
- Maneja formatos latinoamericanos (1.234,56) y anglosajones (1,234.56),
19
+ Convierte cualquier valor numérico, moneda o porcentaje a float o int de forma precisa.
20
+ Maneja formatos latinoamericanos (1.234,56 / 0,056), anglosajones (1,234.56 / 0.056),
21
21
  símbolos ($ / € / USD / CLP / %), y negativos entre paréntesis (100) -> -100.
22
22
 
23
23
  Ejemplos:
24
+ limpiar_numero("0,056") -> 0.056
25
+ limpiar_numero("9,999") -> 9.999
26
+ limpiar_numero("89,949") -> 89.949
27
+ limpiar_numero("0,056%") -> 0.056
24
28
  limpiar_numero("$ 1.250.000") -> 1250000.0
25
- limpiar_numero("18,5 %") -> 18.5
26
- limpiar_numero("(450.50)") -> -450.50
29
+ limpiar_numero("(450,50)") -> -450.50
27
30
  """
28
31
  if pd.isna(val) or val is None:
29
- return np.nan
32
+ return default
30
33
  if isinstance(val, (int, float, np.number)):
31
- return val
34
+ return float(val) if isinstance(val, float) else val
32
35
 
33
36
  s = str(val).strip()
34
37
  if not s:
35
- return np.nan
38
+ return default
36
39
 
37
- # Detectar negativos con paréntesis: (123.45)
40
+ # 1. Detectar negativos con paréntesis: (123.45) o (123,45)
38
41
  es_negativo = False
39
42
  if s.startswith("(") and s.endswith(")"):
40
43
  es_negativo = True
@@ -43,46 +46,48 @@ def limpiar_numero(val: Any) -> Union[float, int, Any]:
43
46
  es_negativo = True
44
47
  s = s[1:].strip()
45
48
 
46
- # Remover símbolos de monedas y palabras comunes
49
+ # 2. Remover símbolos de monedas y texto innecesario
47
50
  s = re.sub(r'[\$\€\£\¥\s]|USD|CLP|EUR|UF', '', s, flags=re.IGNORECASE)
48
51
  # Remover %
49
52
  s = s.replace('%', '').strip()
50
53
 
51
54
  if not s:
52
- return np.nan
55
+ return default
53
56
 
54
- # Determinar si el separador decimal es coma o punto
55
- # Caso 1: Tiene comas y puntos (ej: 1.234.567,89 o 1,234,567.89)
57
+ # 3. Determinar separadores decimales y de miles
58
+ # Caso A: Tiene comas y puntos (ej: 1.234.567,89 o 1,234,567.89)
56
59
  if '.' in s and ',' in s:
57
60
  if s.rfind(',') > s.rfind('.'):
58
- # Formato latino: 1.234,56 -> quitar puntos y coma a punto
61
+ # Formato latino: 1.234.567,89 -> quitar puntos y reemplazar coma por punto
59
62
  s = s.replace('.', '').replace(',', '.')
60
63
  else:
61
- # Formato anglo: 1,234.56 -> quitar comas
64
+ # Formato anglo: 1,234,567.89 -> quitar comas
62
65
  s = s.replace(',', '')
66
+
67
+ # Caso B: Solo tiene coma(s)
63
68
  elif ',' in s:
64
- # Solo tiene comas. Si tiene 1 coma y max 2 decimales al final: 1234,56
65
- partes = s.split(',')
66
- if len(partes) == 2 and len(partes[1]) <= 2:
69
+ num_comas = s.count(',')
70
+ if num_comas == 1:
71
+ # 1 sola coma: SIEMPRE es separador decimal en español (ej: 0,056 | 9,999 | 89,949 | 123,45)
67
72
  s = s.replace(',', '.')
68
73
  else:
69
- # Es separador de miles: 1,000,000
74
+ # Múltiples comas: separador de miles anglosajón (ej: 1,000,000)
70
75
  s = s.replace(',', '')
76
+
77
+ # Caso C: Solo tiene punto(s)
71
78
  elif '.' in s:
72
- # Solo tiene puntos. Si tiene múltiples puntos: 1.000.000
73
- if s.count('.') > 1:
79
+ num_puntos = s.count('.')
80
+ if num_puntos > 1:
81
+ # Múltiples puntos: separador de miles latino (ej: 1.000.000)
74
82
  s = s.replace('.', '')
75
- # Si tiene 1 punto y exactamente 3 dígitos después, puede ser miles (ej: 1.500)
76
- # pero en float estándar suele ser decimal. Lo dejamos a float directo.
77
83
 
78
84
  try:
79
85
  num = float(s)
80
86
  if es_negativo:
81
87
  num = -num
82
- # Si no tiene decimales reales, retornar float o int
83
88
  return num
84
89
  except ValueError:
85
- return val
90
+ return default
86
91
 
87
92
 
88
93
  def limpiar_columnas_numericas(df: pd.DataFrame, columnas: Union[str, List[str]]) -> pd.DataFrame:
@@ -141,6 +146,18 @@ def formato_moneda(val: Any, simbolo: str = "$", decimales: int = 0, separador_m
141
146
  return f"{simbolo} {texto}" if simbolo else texto
142
147
 
143
148
 
149
+ def formato_clp(val: Any, simbolo: str = "$") -> str:
150
+ """
151
+ Formatea un número o texto a Pesos Chilenos (CLP) con separador de miles y sin decimales.
152
+
153
+ Ejemplos:
154
+ formato_clp(1500000) -> "$ 1.500.000"
155
+ formato_clp("1500000") -> "$ 1.500.000"
156
+ formato_clp(250000, simbolo="CLP") -> "CLP 250.000"
157
+ """
158
+ return formato_moneda(val, simbolo=simbolo, decimales=0, separador_miles=".")
159
+
160
+
144
161
  def formato_porcentaje(
145
162
  val: Any,
146
163
  decimales: Optional[int] = None,
@@ -219,10 +236,14 @@ def formatear_dataframe(
219
236
  for col, tipo in reglas.items():
220
237
  if col not in df_out.columns:
221
238
  continue
222
- if tipo == 'moneda':
223
- df_out[col] = df_out[col].apply(lambda x: formato_moneda(x, decimales=0))
239
+ if tipo in ['moneda', 'clp', 'moneda_clp']:
240
+ df_out[col] = df_out[col].apply(lambda x: formato_clp(x))
224
241
  elif tipo == 'moneda_2dec':
225
242
  df_out[col] = df_out[col].apply(lambda x: formato_moneda(x, decimales=2))
243
+ elif tipo == 'usd':
244
+ df_out[col] = df_out[col].apply(lambda x: formato_moneda(x, simbolo="USD $", decimales=2, separador_miles=","))
245
+ elif tipo == 'uf':
246
+ df_out[col] = df_out[col].apply(lambda x: formato_moneda(x, simbolo="UF", decimales=2, separador_miles="."))
226
247
  elif tipo == 'porcentaje':
227
248
  df_out[col] = df_out[col].apply(lambda x: formato_porcentaje(x))
228
249
  elif tipo == 'porcentaje_1dec':
File without changes
File without changes
File without changes