tablas-python 0.1.4__tar.gz → 0.1.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {tablas_python-0.1.4 → tablas_python-0.1.5}/PKG-INFO +1 -1
  2. {tablas_python-0.1.4 → tablas_python-0.1.5}/pyproject.toml +1 -1
  3. {tablas_python-0.1.4 → tablas_python-0.1.5}/tablas_python/__init__.py +1 -1
  4. {tablas_python-0.1.4 → tablas_python-0.1.5}/tablas_python.egg-info/PKG-INFO +1 -1
  5. {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/data_helpers.py +30 -25
  6. {tablas_python-0.1.4 → tablas_python-0.1.5}/LICENSE +0 -0
  7. {tablas_python-0.1.4 → tablas_python-0.1.5}/README.md +0 -0
  8. {tablas_python-0.1.4 → tablas_python-0.1.5}/helpers/__init__.py +0 -0
  9. {tablas_python-0.1.4 → tablas_python-0.1.5}/helpers/display_helper.py +0 -0
  10. {tablas_python-0.1.4 → tablas_python-0.1.5}/helpers/table_manager.py +0 -0
  11. {tablas_python-0.1.4 → tablas_python-0.1.5}/setup.cfg +0 -0
  12. {tablas_python-0.1.4 → tablas_python-0.1.5}/tablas_python/cli.py +0 -0
  13. {tablas_python-0.1.4 → tablas_python-0.1.5}/tablas_python.egg-info/SOURCES.txt +0 -0
  14. {tablas_python-0.1.4 → tablas_python-0.1.5}/tablas_python.egg-info/dependency_links.txt +0 -0
  15. {tablas_python-0.1.4 → tablas_python-0.1.5}/tablas_python.egg-info/entry_points.txt +0 -0
  16. {tablas_python-0.1.4 → tablas_python-0.1.5}/tablas_python.egg-info/requires.txt +0 -0
  17. {tablas_python-0.1.4 → tablas_python-0.1.5}/tablas_python.egg-info/top_level.txt +0 -0
  18. {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/__init__.py +0 -0
  19. {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/batch_processor.py +0 -0
  20. {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/excel_extractor.py +0 -0
  21. {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/excel_writer.py +0 -0
  22. {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/exporter.py +0 -0
  23. {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/file_utils.py +0 -0
  24. {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/pdf_extractor.py +0 -0
  25. {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/sqlite_extractor.py +0 -0
  26. {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/table_cleaner.py +0 -0
  27. {tablas_python-0.1.4 → tablas_python-0.1.5}/utils/validator.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tablas-python
3
- Version: 0.1.4
3
+ Version: 0.1.5
4
4
  Summary: Suite modular para extraer, transformar, conciliar y exportar tablas desde PDF, Excel y SQLite a pandas.
5
5
  Author: Reciba
6
6
  License: MIT
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "tablas-python"
7
- version = "0.1.4"
7
+ version = "0.1.5"
8
8
  description = "Suite modular para extraer, transformar, conciliar y exportar tablas desde PDF, Excel y SQLite a pandas."
9
9
  readme = "README.md"
10
10
  authors = [
@@ -32,7 +32,7 @@ from utils.data_helpers import (
32
32
  conciliar_tablas,
33
33
  )
34
34
 
35
- __version__ = "0.1.4"
35
+ __version__ = "0.1.5"
36
36
 
37
37
  __all__ = [
38
38
  "TableManager",
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tablas-python
3
- Version: 0.1.4
3
+ Version: 0.1.5
4
4
  Summary: Suite modular para extraer, transformar, conciliar y exportar tablas desde PDF, Excel y SQLite a pandas.
5
5
  Author: Reciba
6
6
  License: MIT
@@ -14,27 +14,30 @@ import numpy as np
14
14
  # 1. PARSEO Y LIMPIEZA DE NÚMEROS Y FECHAS
15
15
  # ==========================================
16
16
 
17
- def limpiar_numero(val: Any) -> Union[float, int, Any]:
17
+ def limpiar_numero(val: Any, default: Any = np.nan) -> Union[float, int, Any]:
18
18
  """
19
- Convierte cualquier cadena con formato numérico, moneda o porcentaje a float o int.
20
- Maneja formatos latinoamericanos (1.234,56) y anglosajones (1,234.56),
19
+ Convierte cualquier valor numérico, moneda o porcentaje a float o int de forma precisa.
20
+ Maneja formatos latinoamericanos (1.234,56 / 0,056), anglosajones (1,234.56 / 0.056),
21
21
  símbolos ($ / € / USD / CLP / %), y negativos entre paréntesis (100) -> -100.
22
22
 
23
23
  Ejemplos:
24
+ limpiar_numero("0,056") -> 0.056
25
+ limpiar_numero("9,999") -> 9.999
26
+ limpiar_numero("89,949") -> 89.949
27
+ limpiar_numero("0,056%") -> 0.056
24
28
  limpiar_numero("$ 1.250.000") -> 1250000.0
25
- limpiar_numero("18,5 %") -> 18.5
26
- limpiar_numero("(450.50)") -> -450.50
29
+ limpiar_numero("(450,50)") -> -450.50
27
30
  """
28
31
  if pd.isna(val) or val is None:
29
- return np.nan
32
+ return default
30
33
  if isinstance(val, (int, float, np.number)):
31
- return val
34
+ return float(val) if isinstance(val, float) else val
32
35
 
33
36
  s = str(val).strip()
34
37
  if not s:
35
- return np.nan
38
+ return default
36
39
 
37
- # Detectar negativos con paréntesis: (123.45)
40
+ # 1. Detectar negativos con paréntesis: (123.45) o (123,45)
38
41
  es_negativo = False
39
42
  if s.startswith("(") and s.endswith(")"):
40
43
  es_negativo = True
@@ -43,46 +46,48 @@ def limpiar_numero(val: Any) -> Union[float, int, Any]:
43
46
  es_negativo = True
44
47
  s = s[1:].strip()
45
48
 
46
- # Remover símbolos de monedas y palabras comunes
49
+ # 2. Remover símbolos de monedas y texto innecesario
47
50
  s = re.sub(r'[\$\€\£\¥\s]|USD|CLP|EUR|UF', '', s, flags=re.IGNORECASE)
48
51
  # Remover %
49
52
  s = s.replace('%', '').strip()
50
53
 
51
54
  if not s:
52
- return np.nan
55
+ return default
53
56
 
54
- # Determinar si el separador decimal es coma o punto
55
- # Caso 1: Tiene comas y puntos (ej: 1.234.567,89 o 1,234,567.89)
57
+ # 3. Determinar separadores decimales y de miles
58
+ # Caso A: Tiene comas y puntos (ej: 1.234.567,89 o 1,234,567.89)
56
59
  if '.' in s and ',' in s:
57
60
  if s.rfind(',') > s.rfind('.'):
58
- # Formato latino: 1.234,56 -> quitar puntos y coma a punto
61
+ # Formato latino: 1.234.567,89 -> quitar puntos y reemplazar coma por punto
59
62
  s = s.replace('.', '').replace(',', '.')
60
63
  else:
61
- # Formato anglo: 1,234.56 -> quitar comas
64
+ # Formato anglo: 1,234,567.89 -> quitar comas
62
65
  s = s.replace(',', '')
66
+
67
+ # Caso B: Solo tiene coma(s)
63
68
  elif ',' in s:
64
- # Solo tiene comas. Si tiene 1 coma y max 2 decimales al final: 1234,56
65
- partes = s.split(',')
66
- if len(partes) == 2 and len(partes[1]) <= 2:
69
+ num_comas = s.count(',')
70
+ if num_comas == 1:
71
+ # 1 sola coma: SIEMPRE es separador decimal en español (ej: 0,056 | 9,999 | 89,949 | 123,45)
67
72
  s = s.replace(',', '.')
68
73
  else:
69
- # Es separador de miles: 1,000,000
74
+ # Múltiples comas: separador de miles anglosajón (ej: 1,000,000)
70
75
  s = s.replace(',', '')
76
+
77
+ # Caso C: Solo tiene punto(s)
71
78
  elif '.' in s:
72
- # Solo tiene puntos. Si tiene múltiples puntos: 1.000.000
73
- if s.count('.') > 1:
79
+ num_puntos = s.count('.')
80
+ if num_puntos > 1:
81
+ # Múltiples puntos: separador de miles latino (ej: 1.000.000)
74
82
  s = s.replace('.', '')
75
- # Si tiene 1 punto y exactamente 3 dígitos después, puede ser miles (ej: 1.500)
76
- # pero en float estándar suele ser decimal. Lo dejamos a float directo.
77
83
 
78
84
  try:
79
85
  num = float(s)
80
86
  if es_negativo:
81
87
  num = -num
82
- # Si no tiene decimales reales, retornar float o int
83
88
  return num
84
89
  except ValueError:
85
- return val
90
+ return default
86
91
 
87
92
 
88
93
  def limpiar_columnas_numericas(df: pd.DataFrame, columnas: Union[str, List[str]]) -> pd.DataFrame:
File without changes
File without changes
File without changes