tablas-python 0.1.2__py3-none-any.whl → 0.1.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- helpers/table_manager.py +47 -3
- tablas_python/__init__.py +1 -1
- {tablas_python-0.1.2.dist-info → tablas_python-0.1.4.dist-info}/METADATA +375 -375
- {tablas_python-0.1.2.dist-info → tablas_python-0.1.4.dist-info}/RECORD +11 -11
- utils/data_helpers.py +53 -11
- utils/excel_extractor.py +50 -9
- utils/excel_writer.py +100 -91
- {tablas_python-0.1.2.dist-info → tablas_python-0.1.4.dist-info}/WHEEL +0 -0
- {tablas_python-0.1.2.dist-info → tablas_python-0.1.4.dist-info}/entry_points.txt +0 -0
- {tablas_python-0.1.2.dist-info → tablas_python-0.1.4.dist-info}/licenses/LICENSE +0 -0
- {tablas_python-0.1.2.dist-info → tablas_python-0.1.4.dist-info}/top_level.txt +0 -0
helpers/table_manager.py
CHANGED
|
@@ -58,14 +58,58 @@ class TableManager:
|
|
|
58
58
|
auto_load : bool (por defecto True)
|
|
59
59
|
Carga las tablas automáticamente al instanciar.
|
|
60
60
|
"""
|
|
61
|
+
self.archivo_abierto = archivo_abierto
|
|
62
|
+
self.preferir_xlwings = preferir_xlwings
|
|
63
|
+
self.tables: List[Union[RawTableInfo, RawExcelTableInfo, RawSQLiteTableInfo]] = []
|
|
64
|
+
self.is_open_in_memory = False
|
|
65
|
+
|
|
66
|
+
# 1. Si es Excel y archivo_abierto=True o None, buscar primero en libros abiertos en pantalla (xlwings)
|
|
67
|
+
if self.preferir_xlwings and archivo_abierto is not False:
|
|
68
|
+
try:
|
|
69
|
+
import xlwings as xw
|
|
70
|
+
if not file_path or file_path.lower() in ["", "activo", "active", "libro_activo"]:
|
|
71
|
+
matched_book = xw.books.active if len(xw.books) > 0 else None
|
|
72
|
+
else:
|
|
73
|
+
clean_target = os.path.basename(file_path).strip().lower()
|
|
74
|
+
name_no_ext = os.path.splitext(clean_target)[0]
|
|
75
|
+
matched_book = None
|
|
76
|
+
for b in xw.books:
|
|
77
|
+
b_name = b.name.lower()
|
|
78
|
+
b_name_no_ext = os.path.splitext(b_name)[0]
|
|
79
|
+
b_fullname = getattr(b, 'fullname', '').lower()
|
|
80
|
+
if (clean_target == b_name or
|
|
81
|
+
name_no_ext == b_name_no_ext or
|
|
82
|
+
(b_fullname and clean_target == os.path.basename(b_fullname).lower())):
|
|
83
|
+
matched_book = b
|
|
84
|
+
break
|
|
85
|
+
|
|
86
|
+
if matched_book:
|
|
87
|
+
self.file_path = getattr(matched_book, 'fullname', matched_book.name)
|
|
88
|
+
self.ext = os.path.splitext(matched_book.name)[1].lower() or ".xlsx"
|
|
89
|
+
self.is_open_in_memory = True
|
|
90
|
+
if auto_load:
|
|
91
|
+
self.cargar_tablas()
|
|
92
|
+
return
|
|
93
|
+
except Exception:
|
|
94
|
+
pass
|
|
95
|
+
|
|
96
|
+
if archivo_abierto is True:
|
|
97
|
+
try:
|
|
98
|
+
import xlwings as xw
|
|
99
|
+
abiertos = [b.name for b in xw.books] if len(xw.books) > 0 else []
|
|
100
|
+
except Exception:
|
|
101
|
+
abiertos = []
|
|
102
|
+
raise FileNotFoundError(
|
|
103
|
+
f"No se encontró ningún libro abierto en Excel con el nombre '{file_path}'. "
|
|
104
|
+
f"Libros actualmente abiertos en Excel: {abiertos if abiertos else 'Ninguno (Excel no tiene libros abiertos)'}"
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
# 2. Si no es un libro abierto en memoria, resolver la ruta física en disco
|
|
61
108
|
self.file_path = resolve_file_path(file_path)
|
|
62
109
|
if not os.path.exists(self.file_path):
|
|
63
110
|
raise FileNotFoundError(f"No se encontró el archivo: {self.file_path}")
|
|
64
111
|
|
|
65
112
|
self.ext = os.path.splitext(self.file_path)[1].lower()
|
|
66
|
-
self.archivo_abierto = archivo_abierto
|
|
67
|
-
self.preferir_xlwings = preferir_xlwings
|
|
68
|
-
self.tables: List[Union[RawTableInfo, RawExcelTableInfo, RawSQLiteTableInfo]] = []
|
|
69
113
|
|
|
70
114
|
valid_extensions = ['.pdf', '.xlsx', '.xls', '.xlsm', '.csv', '.db', '.sqlite', '.sqlite3', '.db3']
|
|
71
115
|
if self.ext not in valid_extensions:
|
tablas_python/__init__.py
CHANGED
|
@@ -1,375 +1,375 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: tablas-python
|
|
3
|
-
Version: 0.1.
|
|
4
|
-
Summary: Suite modular para extraer, transformar, conciliar y exportar tablas desde PDF, Excel y SQLite a pandas.
|
|
5
|
-
Author: Reciba
|
|
6
|
-
License: MIT
|
|
7
|
-
Project-URL: Homepage, https://github.com/Reciba/tablas_python
|
|
8
|
-
Project-URL: Repository, https://github.com/Reciba/tablas_python.git
|
|
9
|
-
Project-URL: Issues, https://github.com/Reciba/tablas_python/issues
|
|
10
|
-
Keywords: tables,pdf,excel,xlwings,pdfplumber,sqlite,pandas,data-extraction,vlookup,conciliation
|
|
11
|
-
Classifier: Programming Language :: Python :: 3
|
|
12
|
-
Classifier: Programming Language :: Python :: 3.8
|
|
13
|
-
Classifier: Programming Language :: Python :: 3.9
|
|
14
|
-
Classifier: Programming Language :: Python :: 3.10
|
|
15
|
-
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
-
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
-
Classifier: License :: OSI Approved :: MIT License
|
|
18
|
-
Classifier: Operating System :: OS Independent
|
|
19
|
-
Classifier: Topic :: Office/Business :: Financial :: Spreadsheet
|
|
20
|
-
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
21
|
-
Requires-Python: >=3.8
|
|
22
|
-
Description-Content-Type: text/markdown
|
|
23
|
-
License-File: LICENSE
|
|
24
|
-
Requires-Dist: pandas>=1.5.0
|
|
25
|
-
Requires-Dist: pdfplumber>=0.9.0
|
|
26
|
-
Requires-Dist: openpyxl>=3.1.0
|
|
27
|
-
Requires-Dist: xlwings>=0.30.0
|
|
28
|
-
Requires-Dist: rich>=13.0.0
|
|
29
|
-
Requires-Dist: numpy>=1.20.0
|
|
30
|
-
Provides-Extra: dev
|
|
31
|
-
Requires-Dist: build; extra == "dev"
|
|
32
|
-
Requires-Dist: twine; extra == "dev"
|
|
33
|
-
Requires-Dist: pytest; extra == "dev"
|
|
34
|
-
Requires-Dist: reportlab>=3.6.0; extra == "dev"
|
|
35
|
-
Dynamic: license-file
|
|
36
|
-
|
|
37
|
-
# 📊 Tablas Python - Suite Integral de Procesamiento de Tablas
|
|
38
|
-
|
|
39
|
-
Sistema modular en Python diseñado para **reconocer, extraer, transformar, conciliar, validar y exportar tablas** provenientes de múltiples fuentes:
|
|
40
|
-
- 📄 **Documentos PDF** (usando `pdfplumber` con descarte de filas de basura iniciales)
|
|
41
|
-
- 📊 **Archivos Excel / CSV** (usando `xlwings`, compatible con libros **abiertos** o **cerrados**, tablas oficiales `ListObjects`, celdas de inicio y rangos)
|
|
42
|
-
- 🗄️ **Bases de Datos SQLite** (archivos `.db`, `.sqlite`, `.sqlite3`)
|
|
43
|
-
|
|
44
|
-
Además, incluye un conjunto completo de herramientas para **cálculos rápidos (IVA, márgenes, % participación), formateo de monedas/porcentajes, cruces tipo `BUSCARV`, conciliación automática de tablas, escritura en vivo en Excel, unión masiva de carpetas y reportes de calidad de datos**.
|
|
45
|
-
|
|
46
|
-
---
|
|
47
|
-
|
|
48
|
-
## 📁 Arquitectura del Proyecto
|
|
49
|
-
|
|
50
|
-
El código está estructurado de forma modular y desacoplada por capas:
|
|
51
|
-
|
|
52
|
-
```text
|
|
53
|
-
tablas_python/
|
|
54
|
-
│
|
|
55
|
-
├── main.py # Script principal con todas las utilidades importadas y listas para usar
|
|
56
|
-
├── requirements.txt # Dependencias del proyecto
|
|
57
|
-
├── README.md # Documentación técnica completa
|
|
58
|
-
├── .gitignore # Exclusiones de Git
|
|
59
|
-
│
|
|
60
|
-
├── utils/ # Capa de bajo nivel (lógica desacoplada y reutilizable)
|
|
61
|
-
│ ├── __init__.py # Exportaciones unificadas del paquete utils
|
|
62
|
-
│ ├── file_utils.py # Resolución de rutas (relativas/absolutas) y nombres seguros
|
|
63
|
-
│ ├── pdf_extractor.py # Extracción multipágina en PDF con pdfplumber
|
|
64
|
-
│ ├── excel_extractor.py # Extracción en Excel con xlwings (tablas oficiales, celdas y rangos)
|
|
65
|
-
│ ├── excel_writer.py # Escritura en Excel en vivo o en segundo plano con xlwings
|
|
66
|
-
│ ├── sqlite_extractor.py # Extracción y consultas SQL en SQLite
|
|
67
|
-
│ ├── table_cleaner.py # Limpieza, descarte de basura superior, tipos y normalización
|
|
68
|
-
│ ├── batch_processor.py # Unión masiva de carpetas con múltiples archivos a 1 DataFrame
|
|
69
|
-
│ ├── validator.py # Validación de esquemas, diagnóstico de calidad y duplicados
|
|
70
|
-
│ ├── exporter.py # Exportación individual y por lotes a CSV/Excel (utf-8-sig / sep=';')
|
|
71
|
-
│ └── data_helpers.py # Cálculos: BUSCARV, conciliación, IVA, %, totales, celdas y formato
|
|
72
|
-
│
|
|
73
|
-
├── helpers/ # Capa de fachada y presentación
|
|
74
|
-
│ ├── __init__.py # Exportaciones unificadas del paquete helpers
|
|
75
|
-
│ ├── table_manager.py # TableManager unificado, obtener_tabla y exportar_archivo_a_csv
|
|
76
|
-
│ └── display_helper.py # Visualizador enriquecido en consola con Rich
|
|
77
|
-
│
|
|
78
|
-
├── samples/ # Archivos y generador de prueba
|
|
79
|
-
│ ├── generate_samples.py # Script generador de datos de prueba
|
|
80
|
-
│ ├── ejemplo_facturas.pdf # PDF de muestra con tablas y encabezados desplazados
|
|
81
|
-
│ ├── ejemplo_inventario.xlsx # Excel con tablas oficiales, celdas específicas y múltiples hojas
|
|
82
|
-
│ └── ejemplo_empresa.db # Base SQLite de muestra con tablas 'clientes', 'ventas' y vistas
|
|
83
|
-
│
|
|
84
|
-
└── exports/ # Carpeta por defecto para salidas CSV y Excel
|
|
85
|
-
```
|
|
86
|
-
|
|
87
|
-
---
|
|
88
|
-
|
|
89
|
-
## 🚀 Instalación
|
|
90
|
-
|
|
91
|
-
Clona el repositorio e instala las dependencias:
|
|
92
|
-
|
|
93
|
-
```bash
|
|
94
|
-
git clone https://github.com/Reciba/tablas_python.git
|
|
95
|
-
cd tablas_python
|
|
96
|
-
pip install -r requirements.txt
|
|
97
|
-
```
|
|
98
|
-
|
|
99
|
-
---
|
|
100
|
-
|
|
101
|
-
## 💡 Guía de Uso Completa
|
|
102
|
-
|
|
103
|
-
### 1. Extracción desde PDF (`pdfplumber`)
|
|
104
|
-
|
|
105
|
-
Resuelve el problema donde las filas 1, 2 o 3 contienen títulos, membretes o metadatos irrelevantes y los encabezados reales comienzan más abajo (por ejemplo en la **fila 4**):
|
|
106
|
-
|
|
107
|
-
```python
|
|
108
|
-
from helpers.table_manager import obtener_tabla, TableManager
|
|
109
|
-
|
|
110
|
-
# Modo 1: En una sola línea indicando tabla 3 y fila 4 como encabezado:
|
|
111
|
-
df = obtener_tabla("samples/ejemplo_facturas.pdf", tabla=3, fila_encabezado=4)
|
|
112
|
-
print(df.head())
|
|
113
|
-
|
|
114
|
-
# Modo 2: Con inspección previa usando TableManager:
|
|
115
|
-
manager = TableManager("samples/ejemplo_facturas.pdf")
|
|
116
|
-
manager.resumen() # Muestra cuántas tablas hay y en qué páginas
|
|
117
|
-
manager.ver_crudo(tabla=1) # Muestra las primeras filas numeradas (Fila 1, Fila 2, Fila 3...)
|
|
118
|
-
df_limpio = manager.get_df(tabla=1, fila_encabezado=4, skip_footer=1)
|
|
119
|
-
```
|
|
120
|
-
|
|
121
|
-
---
|
|
122
|
-
|
|
123
|
-
### 2. Extracción desde Excel con `xlwings`
|
|
124
|
-
|
|
125
|
-
Soporta conectarse a archivos **abiertos en pantalla** (sin conflictos de bloqueo) o **cerrados en disco** (en segundo plano):
|
|
126
|
-
|
|
127
|
-
```python
|
|
128
|
-
from helpers.table_manager import obtener_tabla
|
|
129
|
-
|
|
130
|
-
# A) Por Nombre Oficial de Tabla de Excel (ListObject / Tabla con Formato):
|
|
131
|
-
df_stock = obtener_tabla("samples/ejemplo_inventario.xlsx", tabla="TablaStock")
|
|
132
|
-
|
|
133
|
-
# B) Por Celda de Inicio donde parte la tabla (ej. celda C4 en la hoja 'Despacho'):
|
|
134
|
-
df_despacho = obtener_tabla(
|
|
135
|
-
"samples/ejemplo_inventario.xlsx",
|
|
136
|
-
celda_inicio="C4",
|
|
137
|
-
hoja="Despacho"
|
|
138
|
-
)
|
|
139
|
-
|
|
140
|
-
# C) Por Rango Exacto:
|
|
141
|
-
df_rango = obtener_tabla(
|
|
142
|
-
"samples/ejemplo_inventario.xlsx",
|
|
143
|
-
rango="C4:F8",
|
|
144
|
-
hoja="Despacho"
|
|
145
|
-
)
|
|
146
|
-
|
|
147
|
-
# D) Control de archivo abierto/cerrado:
|
|
148
|
-
# archivo_abierto=None (autodetecta), True (fuerza conexión al Excel abierto), False (abre oculto)
|
|
149
|
-
df = obtener_tabla("reporte.xlsx", tabla=1, fila_encabezado=4, archivo_abierto=True)
|
|
150
|
-
```
|
|
151
|
-
|
|
152
|
-
---
|
|
153
|
-
|
|
154
|
-
### 3. Extracción y Consultas en SQLite (`.db`, `.sqlite`)
|
|
155
|
-
|
|
156
|
-
```python
|
|
157
|
-
from helpers.table_manager import TableManager, obtener_tabla
|
|
158
|
-
|
|
159
|
-
# A) Obtener tabla completa por nombre:
|
|
160
|
-
df_clientes = obtener_tabla("samples/ejemplo_empresa.db", tabla="clientes")
|
|
161
|
-
|
|
162
|
-
# B) Ejecutar consultas SQL personalizadas directamente a DataFrame:
|
|
163
|
-
db = TableManager("samples/ejemplo_empresa.db")
|
|
164
|
-
db.resumen() # Lista tablas y vistas
|
|
165
|
-
|
|
166
|
-
df_ventas = db.query("""
|
|
167
|
-
SELECT c.nombre AS cliente, c.ciudad, v.monto_neto, v.fecha
|
|
168
|
-
FROM clientes c
|
|
169
|
-
JOIN ventas v ON c.id_cliente = v.id_cliente
|
|
170
|
-
WHERE v.estado = 'Pagado'
|
|
171
|
-
""")
|
|
172
|
-
```
|
|
173
|
-
|
|
174
|
-
---
|
|
175
|
-
|
|
176
|
-
### 4. Búsqueda de Celdas por Nombre de Fila y Nombre de Columna
|
|
177
|
-
|
|
178
|
-
Puedes acceder a cualquier celda específica mediante nombres sin depender de posiciones numéricas:
|
|
179
|
-
|
|
180
|
-
```python
|
|
181
|
-
from utils.data_helpers import obtener_celda, modificar_celda
|
|
182
|
-
|
|
183
|
-
# A) Usando la función helper obtener_celda:
|
|
184
|
-
precio = obtener_celda(df, fila="PROD-101", columna="Precio Unitario")
|
|
185
|
-
print("Precio PROD-101:", precio)
|
|
186
|
-
|
|
187
|
-
# B) Usando set_index nativo de pandas con .at o .loc:
|
|
188
|
-
df_idx = df.set_index("Codigo")
|
|
189
|
-
precio = df_idx.at["PROD-101", "Precio Unitario"]
|
|
190
|
-
cantidad = df_idx.at["PROD-101", "Cantidad"]
|
|
191
|
-
total = float(precio) * float(cantidad)
|
|
192
|
-
|
|
193
|
-
# C) Modificar una celda por su nombre:
|
|
194
|
-
df_actualizado = modificar_celda(df, fila="PROD-101", columna="Precio Unitario", nuevo_valor=49.90)
|
|
195
|
-
```
|
|
196
|
-
|
|
197
|
-
---
|
|
198
|
-
|
|
199
|
-
### 5. Cruces de Datos con `buscar_v` (BUSCARV / VLOOKUP en 1 línea)
|
|
200
|
-
|
|
201
|
-
Cruza dos DataFrames asociando datos a partir de una clave en común:
|
|
202
|
-
|
|
203
|
-
```python
|
|
204
|
-
from utils.data_helpers import buscar_v
|
|
205
|
-
|
|
206
|
-
# Trae el nombre del cliente desde df_clientes a df_ventas usando 'id_cliente':
|
|
207
|
-
df_ventas["Nombre_Cliente"] = buscar_v(
|
|
208
|
-
df_origen=df_ventas,
|
|
209
|
-
df_destino=df_clientes,
|
|
210
|
-
clave="id_cliente",
|
|
211
|
-
columna_a_traer="nombre"
|
|
212
|
-
)
|
|
213
|
-
```
|
|
214
|
-
|
|
215
|
-
---
|
|
216
|
-
|
|
217
|
-
### 6. Conciliación y Auditoría entre 2 Tablas (`conciliar_tablas`)
|
|
218
|
-
|
|
219
|
-
Compara dos tablas (por ejemplo: extracto bancario vs. registro contable, o inventario físico vs. teórico):
|
|
220
|
-
|
|
221
|
-
```python
|
|
222
|
-
from utils.data_helpers import conciliar_tablas
|
|
223
|
-
|
|
224
|
-
resultado = conciliar_tablas(df_sistema, df_banco, clave="id_transaccion")
|
|
225
|
-
|
|
226
|
-
print("Coincidentes:", resultado["coincidentes"]) # Idénticas en ambas tablas
|
|
227
|
-
print("Diferencias:", resultado["diferencias"]) # Existen en ambas pero con valores distintos
|
|
228
|
-
print("Solo en A:", resultado["solo_en_A"]) # Registros que faltan en B
|
|
229
|
-
print("Solo en B:", resultado["solo_en_B"]) # Registros que faltan en A
|
|
230
|
-
```
|
|
231
|
-
|
|
232
|
-
---
|
|
233
|
-
|
|
234
|
-
### 7. Escritura en Vivo en Excel con `escribir_en_excel`
|
|
235
|
-
|
|
236
|
-
Pega un DataFrame en una celda exacta de una plantilla Excel que tengas **abierta en pantalla** o **cerrada en disco**, conservando fórmulas y formatos:
|
|
237
|
-
|
|
238
|
-
```python
|
|
239
|
-
from utils.excel_writer import escribir_en_excel
|
|
240
|
-
|
|
241
|
-
escribir_en_excel(
|
|
242
|
-
df=df_reporte,
|
|
243
|
-
archivo_excel="plantilla_ventas.xlsx",
|
|
244
|
-
hoja="Resumen",
|
|
245
|
-
celda_inicio="B5",
|
|
246
|
-
archivo_abierto=None # Detecta automáticamente si la ventana de Excel está abierta
|
|
247
|
-
)
|
|
248
|
-
```
|
|
249
|
-
|
|
250
|
-
---
|
|
251
|
-
|
|
252
|
-
### 8. Consolidación Masiva de Carpetas (`unir_archivos_carpeta`)
|
|
253
|
-
|
|
254
|
-
Lee y une decenas de archivos periódicos (facturas PDF, Excels mensuales, etc.) en un solo DataFrame maestro:
|
|
255
|
-
|
|
256
|
-
```python
|
|
257
|
-
from utils.batch_processor import unir_archivos_carpeta
|
|
258
|
-
|
|
259
|
-
df_consolidado = unir_archivos_carpeta(
|
|
260
|
-
carpeta="facturas_2026/",
|
|
261
|
-
extension="pdf",
|
|
262
|
-
tabla=1,
|
|
263
|
-
fila_encabezado=4
|
|
264
|
-
)
|
|
265
|
-
# Agrega automáticamente la columna 'Archivo_Origen' con el nombre de cada archivo
|
|
266
|
-
```
|
|
267
|
-
|
|
268
|
-
---
|
|
269
|
-
|
|
270
|
-
### 9. Calidad y Validación de Datos (`validar_dataframe` y `reporte_calidad`)
|
|
271
|
-
|
|
272
|
-
```python
|
|
273
|
-
from utils.validator import validar_dataframe, reporte_calidad, detectar_duplicados
|
|
274
|
-
|
|
275
|
-
# 1. Diagnóstico completo de nulos, completitud y tipos por columna:
|
|
276
|
-
print(reporte_calidad(df))
|
|
277
|
-
|
|
278
|
-
# 2. Validación de reglas de negocio antes de procesar:
|
|
279
|
-
check = validar_dataframe(
|
|
280
|
-
df,
|
|
281
|
-
columnas_requeridas=["id_venta", "monto_neto", "fecha"],
|
|
282
|
-
no_nulos=["id_venta", "monto_neto"],
|
|
283
|
-
tipos_esperados={"monto_neto": "numeric"}
|
|
284
|
-
)
|
|
285
|
-
|
|
286
|
-
if not check["es_valido"]:
|
|
287
|
-
print("❌ Errores encontrados:", check["errores"])
|
|
288
|
-
|
|
289
|
-
# 3. Detección de registros duplicados:
|
|
290
|
-
duplicados = detectar_duplicados(df, columnas_clave=["id_venta"])
|
|
291
|
-
```
|
|
292
|
-
|
|
293
|
-
---
|
|
294
|
-
|
|
295
|
-
### 10. Cálculos Rápidos y Formato para Reportes
|
|
296
|
-
|
|
297
|
-
```python
|
|
298
|
-
from utils.data_helpers import (
|
|
299
|
-
aplicar_impuesto,
|
|
300
|
-
calcular_participacion,
|
|
301
|
-
calcular_variacion,
|
|
302
|
-
agregar_fila_totales,
|
|
303
|
-
formatear_dataframe,
|
|
304
|
-
formato_moneda,
|
|
305
|
-
formato_porcentaje,
|
|
306
|
-
limpiar_numero
|
|
307
|
-
)
|
|
308
|
-
|
|
309
|
-
# A) Calcular IVA (19%) y Total Bruto:
|
|
310
|
-
df = aplicar_impuesto(df, col_neto="monto_neto", tasa=0.19)
|
|
311
|
-
|
|
312
|
-
# B) Calcular % de participación sobre el total:
|
|
313
|
-
df = calcular_participacion(df, columna_valor="Total Bruto")
|
|
314
|
-
|
|
315
|
-
# C) Agregar fila final con totales:
|
|
316
|
-
df = agregar_fila_totales(df, columnas_sumar=["monto_neto", "IVA (19%)", "Total Bruto"])
|
|
317
|
-
|
|
318
|
-
# D) Formatear números a moneda ($) y porcentaje (%):
|
|
319
|
-
df_formateado = formatear_dataframe(df, {
|
|
320
|
-
"monto_neto": "moneda",
|
|
321
|
-
"IVA (19%)": "moneda",
|
|
322
|
-
"Total Bruto": "moneda",
|
|
323
|
-
"% Participación": "porcentaje"
|
|
324
|
-
})
|
|
325
|
-
```
|
|
326
|
-
|
|
327
|
-
---
|
|
328
|
-
|
|
329
|
-
### 11. Exportación Fácil a CSV y Excel
|
|
330
|
-
|
|
331
|
-
```python
|
|
332
|
-
from utils.exporter import guardar_csv, guardar_excel
|
|
333
|
-
from helpers.table_manager import exportar_archivo_a_csv
|
|
334
|
-
|
|
335
|
-
# A) Guardar DataFrame individual a CSV (optimizado con ';' y utf-8-sig para Excel):
|
|
336
|
-
guardar_csv(df, "exports/mi_tabla.csv")
|
|
337
|
-
|
|
338
|
-
# B) Guardar a Excel:
|
|
339
|
-
guardar_excel(df, "exports/mi_tabla.xlsx")
|
|
340
|
-
|
|
341
|
-
# C) Exportar TODAS las tablas encontradas en un archivo a CSVs independientes:
|
|
342
|
-
archivos = exportar_archivo_a_csv("samples/ejemplo_facturas.pdf", carpeta_salida="exports/pdf_tablas", fila_encabezado=4)
|
|
343
|
-
```
|
|
344
|
-
|
|
345
|
-
---
|
|
346
|
-
|
|
347
|
-
## 🖥️ Ejecución por Consola
|
|
348
|
-
|
|
349
|
-
```bash
|
|
350
|
-
# 1. Ejecutar script principal de demostración:
|
|
351
|
-
python main.py
|
|
352
|
-
|
|
353
|
-
# 2. Modo interactivo en terminal (asistente guiado):
|
|
354
|
-
python main.py -i
|
|
355
|
-
|
|
356
|
-
# 3. Exportar todas las tablas de cualquier archivo por línea de comandos:
|
|
357
|
-
python main.py -f "samples/ejemplo_facturas.pdf" -r 4 --export-all "exports/salida_csv"
|
|
358
|
-
```
|
|
359
|
-
|
|
360
|
-
---
|
|
361
|
-
|
|
362
|
-
## 📦 Tecnologías Utilizadas
|
|
363
|
-
|
|
364
|
-
- **[pandas](https://pandas.pydata.org/)**: Manipulación y estructuras tabulares en DataFrames.
|
|
365
|
-
- **[xlwings](https://docs.xlwings.org/)**: Integración avanzada con Microsoft Excel (archivos abiertos/cerrados, rangos y tablas).
|
|
366
|
-
- **[pdfplumber](https://github.com/jsvine/pdfplumber)**: Extracción precisa de texto y tablas multipágina en PDFs.
|
|
367
|
-
- **[openpyxl](https://openpyxl.readthedocs.io/)**: Soporte nativo y fallback para archivos `.xlsx`.
|
|
368
|
-
- **[reportlab](https://www.reportlab.com/)**: Generación de PDFs de prueba.
|
|
369
|
-
- **[rich](https://github.com/Textualize/rich)**: Visualización y formato de tablas en consola.
|
|
370
|
-
|
|
371
|
-
---
|
|
372
|
-
|
|
373
|
-
## 📄 Licencia
|
|
374
|
-
|
|
375
|
-
Distribuido bajo licencia MIT. Consulta `LICENSE` para más información.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tablas-python
|
|
3
|
+
Version: 0.1.4
|
|
4
|
+
Summary: Suite modular para extraer, transformar, conciliar y exportar tablas desde PDF, Excel y SQLite a pandas.
|
|
5
|
+
Author: Reciba
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/Reciba/tablas_python
|
|
8
|
+
Project-URL: Repository, https://github.com/Reciba/tablas_python.git
|
|
9
|
+
Project-URL: Issues, https://github.com/Reciba/tablas_python/issues
|
|
10
|
+
Keywords: tables,pdf,excel,xlwings,pdfplumber,sqlite,pandas,data-extraction,vlookup,conciliation
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
18
|
+
Classifier: Operating System :: OS Independent
|
|
19
|
+
Classifier: Topic :: Office/Business :: Financial :: Spreadsheet
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
21
|
+
Requires-Python: >=3.8
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
Requires-Dist: pandas>=1.5.0
|
|
25
|
+
Requires-Dist: pdfplumber>=0.9.0
|
|
26
|
+
Requires-Dist: openpyxl>=3.1.0
|
|
27
|
+
Requires-Dist: xlwings>=0.30.0
|
|
28
|
+
Requires-Dist: rich>=13.0.0
|
|
29
|
+
Requires-Dist: numpy>=1.20.0
|
|
30
|
+
Provides-Extra: dev
|
|
31
|
+
Requires-Dist: build; extra == "dev"
|
|
32
|
+
Requires-Dist: twine; extra == "dev"
|
|
33
|
+
Requires-Dist: pytest; extra == "dev"
|
|
34
|
+
Requires-Dist: reportlab>=3.6.0; extra == "dev"
|
|
35
|
+
Dynamic: license-file
|
|
36
|
+
|
|
37
|
+
# 📊 Tablas Python - Suite Integral de Procesamiento de Tablas
|
|
38
|
+
|
|
39
|
+
Sistema modular en Python diseñado para **reconocer, extraer, transformar, conciliar, validar y exportar tablas** provenientes de múltiples fuentes:
|
|
40
|
+
- 📄 **Documentos PDF** (usando `pdfplumber` con descarte de filas de basura iniciales)
|
|
41
|
+
- 📊 **Archivos Excel / CSV** (usando `xlwings`, compatible con libros **abiertos** o **cerrados**, tablas oficiales `ListObjects`, celdas de inicio y rangos)
|
|
42
|
+
- 🗄️ **Bases de Datos SQLite** (archivos `.db`, `.sqlite`, `.sqlite3`)
|
|
43
|
+
|
|
44
|
+
Además, incluye un conjunto completo de herramientas para **cálculos rápidos (IVA, márgenes, % participación), formateo de monedas/porcentajes, cruces tipo `BUSCARV`, conciliación automática de tablas, escritura en vivo en Excel, unión masiva de carpetas y reportes de calidad de datos**.
|
|
45
|
+
|
|
46
|
+
---
|
|
47
|
+
|
|
48
|
+
## 📁 Arquitectura del Proyecto
|
|
49
|
+
|
|
50
|
+
El código está estructurado de forma modular y desacoplada por capas:
|
|
51
|
+
|
|
52
|
+
```text
|
|
53
|
+
tablas_python/
|
|
54
|
+
│
|
|
55
|
+
├── main.py # Script principal con todas las utilidades importadas y listas para usar
|
|
56
|
+
├── requirements.txt # Dependencias del proyecto
|
|
57
|
+
├── README.md # Documentación técnica completa
|
|
58
|
+
├── .gitignore # Exclusiones de Git
|
|
59
|
+
│
|
|
60
|
+
├── utils/ # Capa de bajo nivel (lógica desacoplada y reutilizable)
|
|
61
|
+
│ ├── __init__.py # Exportaciones unificadas del paquete utils
|
|
62
|
+
│ ├── file_utils.py # Resolución de rutas (relativas/absolutas) y nombres seguros
|
|
63
|
+
│ ├── pdf_extractor.py # Extracción multipágina en PDF con pdfplumber
|
|
64
|
+
│ ├── excel_extractor.py # Extracción en Excel con xlwings (tablas oficiales, celdas y rangos)
|
|
65
|
+
│ ├── excel_writer.py # Escritura en Excel en vivo o en segundo plano con xlwings
|
|
66
|
+
│ ├── sqlite_extractor.py # Extracción y consultas SQL en SQLite
|
|
67
|
+
│ ├── table_cleaner.py # Limpieza, descarte de basura superior, tipos y normalización
|
|
68
|
+
│ ├── batch_processor.py # Unión masiva de carpetas con múltiples archivos a 1 DataFrame
|
|
69
|
+
│ ├── validator.py # Validación de esquemas, diagnóstico de calidad y duplicados
|
|
70
|
+
│ ├── exporter.py # Exportación individual y por lotes a CSV/Excel (utf-8-sig / sep=';')
|
|
71
|
+
│ └── data_helpers.py # Cálculos: BUSCARV, conciliación, IVA, %, totales, celdas y formato
|
|
72
|
+
│
|
|
73
|
+
├── helpers/ # Capa de fachada y presentación
|
|
74
|
+
│ ├── __init__.py # Exportaciones unificadas del paquete helpers
|
|
75
|
+
│ ├── table_manager.py # TableManager unificado, obtener_tabla y exportar_archivo_a_csv
|
|
76
|
+
│ └── display_helper.py # Visualizador enriquecido en consola con Rich
|
|
77
|
+
│
|
|
78
|
+
├── samples/ # Archivos y generador de prueba
|
|
79
|
+
│ ├── generate_samples.py # Script generador de datos de prueba
|
|
80
|
+
│ ├── ejemplo_facturas.pdf # PDF de muestra con tablas y encabezados desplazados
|
|
81
|
+
│ ├── ejemplo_inventario.xlsx # Excel con tablas oficiales, celdas específicas y múltiples hojas
|
|
82
|
+
│ └── ejemplo_empresa.db # Base SQLite de muestra con tablas 'clientes', 'ventas' y vistas
|
|
83
|
+
│
|
|
84
|
+
└── exports/ # Carpeta por defecto para salidas CSV y Excel
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
---
|
|
88
|
+
|
|
89
|
+
## 🚀 Instalación
|
|
90
|
+
|
|
91
|
+
Clona el repositorio e instala las dependencias:
|
|
92
|
+
|
|
93
|
+
```bash
|
|
94
|
+
git clone https://github.com/Reciba/tablas_python.git
|
|
95
|
+
cd tablas_python
|
|
96
|
+
pip install -r requirements.txt
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
---
|
|
100
|
+
|
|
101
|
+
## 💡 Guía de Uso Completa
|
|
102
|
+
|
|
103
|
+
### 1. Extracción desde PDF (`pdfplumber`)
|
|
104
|
+
|
|
105
|
+
Resuelve el problema donde las filas 1, 2 o 3 contienen títulos, membretes o metadatos irrelevantes y los encabezados reales comienzan más abajo (por ejemplo en la **fila 4**):
|
|
106
|
+
|
|
107
|
+
```python
|
|
108
|
+
from helpers.table_manager import obtener_tabla, TableManager
|
|
109
|
+
|
|
110
|
+
# Modo 1: En una sola línea indicando tabla 3 y fila 4 como encabezado:
|
|
111
|
+
df = obtener_tabla("samples/ejemplo_facturas.pdf", tabla=3, fila_encabezado=4)
|
|
112
|
+
print(df.head())
|
|
113
|
+
|
|
114
|
+
# Modo 2: Con inspección previa usando TableManager:
|
|
115
|
+
manager = TableManager("samples/ejemplo_facturas.pdf")
|
|
116
|
+
manager.resumen() # Muestra cuántas tablas hay y en qué páginas
|
|
117
|
+
manager.ver_crudo(tabla=1) # Muestra las primeras filas numeradas (Fila 1, Fila 2, Fila 3...)
|
|
118
|
+
df_limpio = manager.get_df(tabla=1, fila_encabezado=4, skip_footer=1)
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
---
|
|
122
|
+
|
|
123
|
+
### 2. Extracción desde Excel con `xlwings`
|
|
124
|
+
|
|
125
|
+
Soporta conectarse a archivos **abiertos en pantalla** (sin conflictos de bloqueo) o **cerrados en disco** (en segundo plano):
|
|
126
|
+
|
|
127
|
+
```python
|
|
128
|
+
from helpers.table_manager import obtener_tabla
|
|
129
|
+
|
|
130
|
+
# A) Por Nombre Oficial de Tabla de Excel (ListObject / Tabla con Formato):
|
|
131
|
+
df_stock = obtener_tabla("samples/ejemplo_inventario.xlsx", tabla="TablaStock")
|
|
132
|
+
|
|
133
|
+
# B) Por Celda de Inicio donde parte la tabla (ej. celda C4 en la hoja 'Despacho'):
|
|
134
|
+
df_despacho = obtener_tabla(
|
|
135
|
+
"samples/ejemplo_inventario.xlsx",
|
|
136
|
+
celda_inicio="C4",
|
|
137
|
+
hoja="Despacho"
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
# C) Por Rango Exacto:
|
|
141
|
+
df_rango = obtener_tabla(
|
|
142
|
+
"samples/ejemplo_inventario.xlsx",
|
|
143
|
+
rango="C4:F8",
|
|
144
|
+
hoja="Despacho"
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
# D) Control de archivo abierto/cerrado:
|
|
148
|
+
# archivo_abierto=None (autodetecta), True (fuerza conexión al Excel abierto), False (abre oculto)
|
|
149
|
+
df = obtener_tabla("reporte.xlsx", tabla=1, fila_encabezado=4, archivo_abierto=True)
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
---
|
|
153
|
+
|
|
154
|
+
### 3. Extracción y Consultas en SQLite (`.db`, `.sqlite`)
|
|
155
|
+
|
|
156
|
+
```python
|
|
157
|
+
from helpers.table_manager import TableManager, obtener_tabla
|
|
158
|
+
|
|
159
|
+
# A) Obtener tabla completa por nombre:
|
|
160
|
+
df_clientes = obtener_tabla("samples/ejemplo_empresa.db", tabla="clientes")
|
|
161
|
+
|
|
162
|
+
# B) Ejecutar consultas SQL personalizadas directamente a DataFrame:
|
|
163
|
+
db = TableManager("samples/ejemplo_empresa.db")
|
|
164
|
+
db.resumen() # Lista tablas y vistas
|
|
165
|
+
|
|
166
|
+
df_ventas = db.query("""
|
|
167
|
+
SELECT c.nombre AS cliente, c.ciudad, v.monto_neto, v.fecha
|
|
168
|
+
FROM clientes c
|
|
169
|
+
JOIN ventas v ON c.id_cliente = v.id_cliente
|
|
170
|
+
WHERE v.estado = 'Pagado'
|
|
171
|
+
""")
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
---
|
|
175
|
+
|
|
176
|
+
### 4. Búsqueda de Celdas por Nombre de Fila y Nombre de Columna
|
|
177
|
+
|
|
178
|
+
Puedes acceder a cualquier celda específica mediante nombres sin depender de posiciones numéricas:
|
|
179
|
+
|
|
180
|
+
```python
|
|
181
|
+
from utils.data_helpers import obtener_celda, modificar_celda
|
|
182
|
+
|
|
183
|
+
# A) Usando la función helper obtener_celda:
|
|
184
|
+
precio = obtener_celda(df, fila="PROD-101", columna="Precio Unitario")
|
|
185
|
+
print("Precio PROD-101:", precio)
|
|
186
|
+
|
|
187
|
+
# B) Usando set_index nativo de pandas con .at o .loc:
|
|
188
|
+
df_idx = df.set_index("Codigo")
|
|
189
|
+
precio = df_idx.at["PROD-101", "Precio Unitario"]
|
|
190
|
+
cantidad = df_idx.at["PROD-101", "Cantidad"]
|
|
191
|
+
total = float(precio) * float(cantidad)
|
|
192
|
+
|
|
193
|
+
# C) Modificar una celda por su nombre:
|
|
194
|
+
df_actualizado = modificar_celda(df, fila="PROD-101", columna="Precio Unitario", nuevo_valor=49.90)
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
---
|
|
198
|
+
|
|
199
|
+
### 5. Cruces de Datos con `buscar_v` (BUSCARV / VLOOKUP en 1 línea)
|
|
200
|
+
|
|
201
|
+
Cruza dos DataFrames asociando datos a partir de una clave en común:
|
|
202
|
+
|
|
203
|
+
```python
|
|
204
|
+
from utils.data_helpers import buscar_v
|
|
205
|
+
|
|
206
|
+
# Trae el nombre del cliente desde df_clientes a df_ventas usando 'id_cliente':
|
|
207
|
+
df_ventas["Nombre_Cliente"] = buscar_v(
|
|
208
|
+
df_origen=df_ventas,
|
|
209
|
+
df_destino=df_clientes,
|
|
210
|
+
clave="id_cliente",
|
|
211
|
+
columna_a_traer="nombre"
|
|
212
|
+
)
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
---
|
|
216
|
+
|
|
217
|
+
### 6. Conciliación y Auditoría entre 2 Tablas (`conciliar_tablas`)
|
|
218
|
+
|
|
219
|
+
Compara dos tablas (por ejemplo: extracto bancario vs. registro contable, o inventario físico vs. teórico):
|
|
220
|
+
|
|
221
|
+
```python
|
|
222
|
+
from utils.data_helpers import conciliar_tablas
|
|
223
|
+
|
|
224
|
+
resultado = conciliar_tablas(df_sistema, df_banco, clave="id_transaccion")
|
|
225
|
+
|
|
226
|
+
print("Coincidentes:", resultado["coincidentes"]) # Idénticas en ambas tablas
|
|
227
|
+
print("Diferencias:", resultado["diferencias"]) # Existen en ambas pero con valores distintos
|
|
228
|
+
print("Solo en A:", resultado["solo_en_A"]) # Registros que faltan en B
|
|
229
|
+
print("Solo en B:", resultado["solo_en_B"]) # Registros que faltan en A
|
|
230
|
+
```
|
|
231
|
+
|
|
232
|
+
---
|
|
233
|
+
|
|
234
|
+
### 7. Escritura en Vivo en Excel con `escribir_en_excel`
|
|
235
|
+
|
|
236
|
+
Pega un DataFrame en una celda exacta de una plantilla Excel que tengas **abierta en pantalla** o **cerrada en disco**, conservando fórmulas y formatos:
|
|
237
|
+
|
|
238
|
+
```python
|
|
239
|
+
from utils.excel_writer import escribir_en_excel
|
|
240
|
+
|
|
241
|
+
escribir_en_excel(
|
|
242
|
+
df=df_reporte,
|
|
243
|
+
archivo_excel="plantilla_ventas.xlsx",
|
|
244
|
+
hoja="Resumen",
|
|
245
|
+
celda_inicio="B5",
|
|
246
|
+
archivo_abierto=None # Detecta automáticamente si la ventana de Excel está abierta
|
|
247
|
+
)
|
|
248
|
+
```
|
|
249
|
+
|
|
250
|
+
---
|
|
251
|
+
|
|
252
|
+
### 8. Consolidación Masiva de Carpetas (`unir_archivos_carpeta`)
|
|
253
|
+
|
|
254
|
+
Lee y une decenas de archivos periódicos (facturas PDF, Excels mensuales, etc.) en un solo DataFrame maestro:
|
|
255
|
+
|
|
256
|
+
```python
|
|
257
|
+
from utils.batch_processor import unir_archivos_carpeta
|
|
258
|
+
|
|
259
|
+
df_consolidado = unir_archivos_carpeta(
|
|
260
|
+
carpeta="facturas_2026/",
|
|
261
|
+
extension="pdf",
|
|
262
|
+
tabla=1,
|
|
263
|
+
fila_encabezado=4
|
|
264
|
+
)
|
|
265
|
+
# Agrega automáticamente la columna 'Archivo_Origen' con el nombre de cada archivo
|
|
266
|
+
```
|
|
267
|
+
|
|
268
|
+
---
|
|
269
|
+
|
|
270
|
+
### 9. Calidad y Validación de Datos (`validar_dataframe` y `reporte_calidad`)
|
|
271
|
+
|
|
272
|
+
```python
|
|
273
|
+
from utils.validator import validar_dataframe, reporte_calidad, detectar_duplicados
|
|
274
|
+
|
|
275
|
+
# 1. Diagnóstico completo de nulos, completitud y tipos por columna:
|
|
276
|
+
print(reporte_calidad(df))
|
|
277
|
+
|
|
278
|
+
# 2. Validación de reglas de negocio antes de procesar:
|
|
279
|
+
check = validar_dataframe(
|
|
280
|
+
df,
|
|
281
|
+
columnas_requeridas=["id_venta", "monto_neto", "fecha"],
|
|
282
|
+
no_nulos=["id_venta", "monto_neto"],
|
|
283
|
+
tipos_esperados={"monto_neto": "numeric"}
|
|
284
|
+
)
|
|
285
|
+
|
|
286
|
+
if not check["es_valido"]:
|
|
287
|
+
print("❌ Errores encontrados:", check["errores"])
|
|
288
|
+
|
|
289
|
+
# 3. Detección de registros duplicados:
|
|
290
|
+
duplicados = detectar_duplicados(df, columnas_clave=["id_venta"])
|
|
291
|
+
```
|
|
292
|
+
|
|
293
|
+
---
|
|
294
|
+
|
|
295
|
+
### 10. Cálculos Rápidos y Formato para Reportes
|
|
296
|
+
|
|
297
|
+
```python
|
|
298
|
+
from utils.data_helpers import (
|
|
299
|
+
aplicar_impuesto,
|
|
300
|
+
calcular_participacion,
|
|
301
|
+
calcular_variacion,
|
|
302
|
+
agregar_fila_totales,
|
|
303
|
+
formatear_dataframe,
|
|
304
|
+
formato_moneda,
|
|
305
|
+
formato_porcentaje,
|
|
306
|
+
limpiar_numero
|
|
307
|
+
)
|
|
308
|
+
|
|
309
|
+
# A) Calcular IVA (19%) y Total Bruto:
|
|
310
|
+
df = aplicar_impuesto(df, col_neto="monto_neto", tasa=0.19)
|
|
311
|
+
|
|
312
|
+
# B) Calcular % de participación sobre el total:
|
|
313
|
+
df = calcular_participacion(df, columna_valor="Total Bruto")
|
|
314
|
+
|
|
315
|
+
# C) Agregar fila final con totales:
|
|
316
|
+
df = agregar_fila_totales(df, columnas_sumar=["monto_neto", "IVA (19%)", "Total Bruto"])
|
|
317
|
+
|
|
318
|
+
# D) Formatear números a moneda ($) y porcentaje (%):
|
|
319
|
+
df_formateado = formatear_dataframe(df, {
|
|
320
|
+
"monto_neto": "moneda",
|
|
321
|
+
"IVA (19%)": "moneda",
|
|
322
|
+
"Total Bruto": "moneda",
|
|
323
|
+
"% Participación": "porcentaje"
|
|
324
|
+
})
|
|
325
|
+
```
|
|
326
|
+
|
|
327
|
+
---
|
|
328
|
+
|
|
329
|
+
### 11. Exportación Fácil a CSV y Excel
|
|
330
|
+
|
|
331
|
+
```python
|
|
332
|
+
from utils.exporter import guardar_csv, guardar_excel
|
|
333
|
+
from helpers.table_manager import exportar_archivo_a_csv
|
|
334
|
+
|
|
335
|
+
# A) Guardar DataFrame individual a CSV (optimizado con ';' y utf-8-sig para Excel):
|
|
336
|
+
guardar_csv(df, "exports/mi_tabla.csv")
|
|
337
|
+
|
|
338
|
+
# B) Guardar a Excel:
|
|
339
|
+
guardar_excel(df, "exports/mi_tabla.xlsx")
|
|
340
|
+
|
|
341
|
+
# C) Exportar TODAS las tablas encontradas en un archivo a CSVs independientes:
|
|
342
|
+
archivos = exportar_archivo_a_csv("samples/ejemplo_facturas.pdf", carpeta_salida="exports/pdf_tablas", fila_encabezado=4)
|
|
343
|
+
```
|
|
344
|
+
|
|
345
|
+
---
|
|
346
|
+
|
|
347
|
+
## 🖥️ Ejecución por Consola
|
|
348
|
+
|
|
349
|
+
```bash
|
|
350
|
+
# 1. Ejecutar script principal de demostración:
|
|
351
|
+
python main.py
|
|
352
|
+
|
|
353
|
+
# 2. Modo interactivo en terminal (asistente guiado):
|
|
354
|
+
python main.py -i
|
|
355
|
+
|
|
356
|
+
# 3. Exportar todas las tablas de cualquier archivo por línea de comandos:
|
|
357
|
+
python main.py -f "samples/ejemplo_facturas.pdf" -r 4 --export-all "exports/salida_csv"
|
|
358
|
+
```
|
|
359
|
+
|
|
360
|
+
---
|
|
361
|
+
|
|
362
|
+
## 📦 Tecnologías Utilizadas
|
|
363
|
+
|
|
364
|
+
- **[pandas](https://pandas.pydata.org/)**: Manipulación y estructuras tabulares en DataFrames.
|
|
365
|
+
- **[xlwings](https://docs.xlwings.org/)**: Integración avanzada con Microsoft Excel (archivos abiertos/cerrados, rangos y tablas).
|
|
366
|
+
- **[pdfplumber](https://github.com/jsvine/pdfplumber)**: Extracción precisa de texto y tablas multipágina en PDFs.
|
|
367
|
+
- **[openpyxl](https://openpyxl.readthedocs.io/)**: Soporte nativo y fallback para archivos `.xlsx`.
|
|
368
|
+
- **[reportlab](https://www.reportlab.com/)**: Generación de PDFs de prueba.
|
|
369
|
+
- **[rich](https://github.com/Textualize/rich)**: Visualización y formato de tablas en consola.
|
|
370
|
+
|
|
371
|
+
---
|
|
372
|
+
|
|
373
|
+
## 📄 Licencia
|
|
374
|
+
|
|
375
|
+
Distribuido bajo licencia MIT. Consulta `LICENSE` para más información.
|
|
@@ -1,22 +1,22 @@
|
|
|
1
1
|
helpers/__init__.py,sha256=hopndlywciR50FKi6W6Lo_cAOP_YyPsQV3xj8hfkH_Y,899
|
|
2
2
|
helpers/display_helper.py,sha256=i0--Txkjc9__IIKwdkmpx98vzwM_Sekr0RzxvVGEPYA,7749
|
|
3
|
-
helpers/table_manager.py,sha256=
|
|
4
|
-
tablas_python/__init__.py,sha256=
|
|
3
|
+
helpers/table_manager.py,sha256=85OZqMdDqgBBJaln2ehb3-t7RWS1Gowo6aud_zqVHAo,19063
|
|
4
|
+
tablas_python/__init__.py,sha256=xVyYUKPzia_CaMiVFM_WxtIBx-8Y06ltIAWFD0U-cVY,1771
|
|
5
5
|
tablas_python/cli.py,sha256=3BModRFfT8WFkbCSxM9cF_oBK87FetPMGm5F0GTZGNo,1957
|
|
6
|
-
tablas_python-0.1.
|
|
6
|
+
tablas_python-0.1.4.dist-info/licenses/LICENSE,sha256=DQi0EoD04d2_-pnEKrC4WmcmgR6O0WQcJdvma9ILoFQ,1063
|
|
7
7
|
utils/__init__.py,sha256=duEZSrdPZggUMdpJqJsNMQ5aLRaFcONBvhsdJuhCToo,2004
|
|
8
8
|
utils/batch_processor.py,sha256=rE59qkb2sp5J0E5NU77U-CF82Aa-ukrjN2druwgS2Uk,4162
|
|
9
|
-
utils/data_helpers.py,sha256=
|
|
10
|
-
utils/excel_extractor.py,sha256=
|
|
11
|
-
utils/excel_writer.py,sha256=
|
|
9
|
+
utils/data_helpers.py,sha256=U6RTiJd0NpqYtk8kQa37jaJBMCay0CUnduREjaoPLM0,23526
|
|
10
|
+
utils/excel_extractor.py,sha256=0WUhkV5eu9uUusU8zSHQB1ujFG7eQVqI-go4zqxmPFU,18364
|
|
11
|
+
utils/excel_writer.py,sha256=zH_Mtu3Cr_ND_wSHKFyvzH3j16gjVoANpM3dVFXKpr0,6991
|
|
12
12
|
utils/exporter.py,sha256=Oo9Y6dXTg_7t6n824xoxv8d9yuhdHZNWfwLLXvSvtMw,4661
|
|
13
13
|
utils/file_utils.py,sha256=Et3fEaRn70xxpIj5oz84N90ocACK2KkCUr1fO3h6Lgo,2302
|
|
14
14
|
utils/pdf_extractor.py,sha256=3JRMqnEfhBVaUOeYfiNH1rReC4EoSbzDwk919x6647w,4768
|
|
15
15
|
utils/sqlite_extractor.py,sha256=mvKlKY-nsd5cV73nn3xFoqg5dgagPJceRDJfPLG0cbs,5033
|
|
16
16
|
utils/table_cleaner.py,sha256=ncyiunqpVdKParFcRKwQbCqyKZgpyKRN0aUM0Oug9wU,10465
|
|
17
17
|
utils/validator.py,sha256=0UywK_D510ZQCcSkhFRKhEuw48788NM_RFVC-bIFZkE,5837
|
|
18
|
-
tablas_python-0.1.
|
|
19
|
-
tablas_python-0.1.
|
|
20
|
-
tablas_python-0.1.
|
|
21
|
-
tablas_python-0.1.
|
|
22
|
-
tablas_python-0.1.
|
|
18
|
+
tablas_python-0.1.4.dist-info/METADATA,sha256=jpBHw3DHlXyFY2qoyOofl65SS0PTXuvRATwthrL0qZI,13611
|
|
19
|
+
tablas_python-0.1.4.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
20
|
+
tablas_python-0.1.4.dist-info/entry_points.txt,sha256=l29nooItcUpsFzN5LmxM1b-j0qih61vJB9tX8KUwlMg,89
|
|
21
|
+
tablas_python-0.1.4.dist-info/top_level.txt,sha256=Foud-Tbf5pviUJmp7wc3v24IjIyypkvbcZt7HkrWvss,28
|
|
22
|
+
tablas_python-0.1.4.dist-info/RECORD,,
|
utils/data_helpers.py
CHANGED
|
@@ -141,22 +141,56 @@ def formato_moneda(val: Any, simbolo: str = "$", decimales: int = 0, separador_m
|
|
|
141
141
|
return f"{simbolo} {texto}" if simbolo else texto
|
|
142
142
|
|
|
143
143
|
|
|
144
|
-
def formato_porcentaje(
|
|
144
|
+
def formato_porcentaje(
|
|
145
|
+
val: Any,
|
|
146
|
+
decimales: Optional[int] = None,
|
|
147
|
+
multiplicar_por_100: bool = False,
|
|
148
|
+
separador_decimal: str = ","
|
|
149
|
+
) -> str:
|
|
145
150
|
"""
|
|
146
|
-
Formatea un valor a porcentaje.
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
151
|
+
Formatea un valor a porcentaje de manera exacta.
|
|
152
|
+
|
|
153
|
+
A diferencia de versiones anteriores, NO multiplica arbitrariamente por 100
|
|
154
|
+
a menos que se indique explícitamente con `multiplicar_por_100=True`.
|
|
155
|
+
Por lo tanto, 0.056 se muestra como 0,056% y 5.56 se muestra como 5,56%.
|
|
156
|
+
|
|
157
|
+
Ejemplos:
|
|
158
|
+
formato_porcentaje(5.56) -> "5,56%"
|
|
159
|
+
formato_porcentaje(0.056) -> "0,056%"
|
|
160
|
+
formato_porcentaje("0.056%") -> "0,056%"
|
|
161
|
+
formato_porcentaje(0.056, multiplicar_por_100=True) -> "5,6%"
|
|
162
|
+
formato_porcentaje(15.42, decimales=1) -> "15,4%"
|
|
150
163
|
"""
|
|
164
|
+
if pd.isna(val) or val is None or str(val).strip() == "":
|
|
165
|
+
return ""
|
|
166
|
+
|
|
167
|
+
s_orig = str(val).strip()
|
|
168
|
+
tenia_simbolo_pct = "%" in s_orig
|
|
169
|
+
|
|
151
170
|
num = limpiar_numero(val)
|
|
152
171
|
if pd.isna(num) or not isinstance(num, (int, float)):
|
|
153
|
-
return
|
|
154
|
-
|
|
155
|
-
#
|
|
156
|
-
if
|
|
172
|
+
return s_orig
|
|
173
|
+
|
|
174
|
+
# Solo multiplicar por 100 si el usuario lo pide explícitamente y el valor NO traía ya el '%'
|
|
175
|
+
if multiplicar_por_100 and not tenia_simbolo_pct:
|
|
157
176
|
num = num * 100
|
|
158
177
|
|
|
159
|
-
|
|
178
|
+
if decimales is not None:
|
|
179
|
+
fmt_str = f"{{:.{decimales}f}}"
|
|
180
|
+
txt_num = fmt_str.format(num)
|
|
181
|
+
else:
|
|
182
|
+
# Conservar los decimales significativos del número sin truncar
|
|
183
|
+
if isinstance(num, int) or (isinstance(num, float) and num.is_integer()):
|
|
184
|
+
txt_num = str(int(num))
|
|
185
|
+
else:
|
|
186
|
+
txt_num = f"{num:.6f}".rstrip('0').rstrip('.')
|
|
187
|
+
|
|
188
|
+
if separador_decimal == ",":
|
|
189
|
+
txt_num = txt_num.replace(".", ",")
|
|
190
|
+
else:
|
|
191
|
+
txt_num = txt_num.replace(",", ".")
|
|
192
|
+
|
|
193
|
+
return f"{txt_num}%"
|
|
160
194
|
|
|
161
195
|
|
|
162
196
|
def formato_miles(val: Any, decimales: int = 0) -> str:
|
|
@@ -176,6 +210,8 @@ def formatear_dataframe(
|
|
|
176
210
|
'Ventas': 'moneda',
|
|
177
211
|
'Precio': 'moneda_2dec',
|
|
178
212
|
'Margen': 'porcentaje',
|
|
213
|
+
'Participacion': 'porcentaje_1dec',
|
|
214
|
+
'Ratio': 'ratio_pct',
|
|
179
215
|
'Cantidad': 'miles'
|
|
180
216
|
})
|
|
181
217
|
"""
|
|
@@ -188,7 +224,13 @@ def formatear_dataframe(
|
|
|
188
224
|
elif tipo == 'moneda_2dec':
|
|
189
225
|
df_out[col] = df_out[col].apply(lambda x: formato_moneda(x, decimales=2))
|
|
190
226
|
elif tipo == 'porcentaje':
|
|
191
|
-
df_out[col] = df_out[col].apply(formato_porcentaje)
|
|
227
|
+
df_out[col] = df_out[col].apply(lambda x: formato_porcentaje(x))
|
|
228
|
+
elif tipo == 'porcentaje_1dec':
|
|
229
|
+
df_out[col] = df_out[col].apply(lambda x: formato_porcentaje(x, decimales=1))
|
|
230
|
+
elif tipo == 'porcentaje_2dec':
|
|
231
|
+
df_out[col] = df_out[col].apply(lambda x: formato_porcentaje(x, decimales=2))
|
|
232
|
+
elif tipo == 'ratio_pct':
|
|
233
|
+
df_out[col] = df_out[col].apply(lambda x: formato_porcentaje(x, multiplicar_por_100=True, decimales=1))
|
|
192
234
|
elif tipo == 'miles':
|
|
193
235
|
df_out[col] = df_out[col].apply(formato_miles)
|
|
194
236
|
return df_out
|
utils/excel_extractor.py
CHANGED
|
@@ -63,25 +63,66 @@ class ExcelTableExtractor:
|
|
|
63
63
|
archivo_abierto: Optional[bool] = None,
|
|
64
64
|
preferir_xlwings: bool = True
|
|
65
65
|
):
|
|
66
|
+
self.archivo_abierto = archivo_abierto
|
|
67
|
+
self.preferir_xlwings = preferir_xlwings and HAS_XLWINGS
|
|
68
|
+
self.book_name = os.path.basename(file_path) if file_path else ""
|
|
69
|
+
self.is_open_in_memory = False
|
|
70
|
+
|
|
71
|
+
# 1. Si el usuario indicó archivo_abierto=True o None, buscar primero en libros abiertos de xlwings
|
|
72
|
+
if self.preferir_xlwings and archivo_abierto is not False:
|
|
73
|
+
matched_book = self._find_open_workbook_in_excel(file_path)
|
|
74
|
+
if matched_book:
|
|
75
|
+
self.book_name = matched_book.name
|
|
76
|
+
self.file_path = getattr(matched_book, 'fullname', matched_book.name)
|
|
77
|
+
self.ext = os.path.splitext(matched_book.name)[1].lower() or ".xlsx"
|
|
78
|
+
self.is_open_in_memory = True
|
|
79
|
+
return
|
|
80
|
+
|
|
81
|
+
# Si forzó archivo_abierto=True pero no se encontró libro abierto en Excel
|
|
82
|
+
if archivo_abierto is True and self.preferir_xlwings:
|
|
83
|
+
abiertos = [b.name for b in xw.books] if HAS_XLWINGS and len(xw.books) > 0 else []
|
|
84
|
+
raise FileNotFoundError(
|
|
85
|
+
f"No se encontró ningún libro abierto en Excel con el nombre '{file_path}'. "
|
|
86
|
+
f"Libros actualmente abiertos en Excel: {abiertos if abiertos else 'Ninguno (Excel no tiene libros abiertos)'}"
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
# 2. Si es un archivo cerrado o en disco, resolver la ruta física
|
|
66
90
|
self.file_path = resolve_file_path(file_path)
|
|
67
91
|
if not os.path.exists(self.file_path):
|
|
68
92
|
raise FileNotFoundError(f"No se encontró el archivo Excel en: {self.file_path}")
|
|
69
93
|
|
|
70
94
|
self.ext = os.path.splitext(self.file_path)[1].lower()
|
|
71
|
-
self.archivo_abierto = archivo_abierto
|
|
72
|
-
self.preferir_xlwings = preferir_xlwings and HAS_XLWINGS
|
|
73
95
|
|
|
74
|
-
def
|
|
75
|
-
"""
|
|
96
|
+
def _find_open_workbook_in_excel(self, target: str) -> Optional[Any]:
|
|
97
|
+
"""Busca y retorna el objeto Book de xlwings si está abierto en memoria."""
|
|
76
98
|
if not HAS_XLWINGS:
|
|
77
|
-
return
|
|
99
|
+
return None
|
|
78
100
|
try:
|
|
101
|
+
if not target or target.lower() in ["", "activo", "active", "libro_activo", "hoja_activa"]:
|
|
102
|
+
if len(xw.books) > 0:
|
|
103
|
+
return xw.books.active
|
|
104
|
+
return None
|
|
105
|
+
|
|
106
|
+
clean_target = os.path.basename(target).strip().lower()
|
|
107
|
+
name_no_ext = os.path.splitext(clean_target)[0]
|
|
108
|
+
|
|
79
109
|
for book in xw.books:
|
|
80
|
-
|
|
81
|
-
|
|
110
|
+
b_name = book.name.lower()
|
|
111
|
+
b_name_no_ext = os.path.splitext(b_name)[0]
|
|
112
|
+
b_fullname = getattr(book, 'fullname', '').lower()
|
|
113
|
+
|
|
114
|
+
# Comparar con nombre completo, sin extensión o fullname
|
|
115
|
+
if (clean_target == b_name or
|
|
116
|
+
name_no_ext == b_name_no_ext or
|
|
117
|
+
(b_fullname and clean_target == os.path.basename(b_fullname).lower())):
|
|
118
|
+
return book
|
|
82
119
|
except Exception:
|
|
83
|
-
return
|
|
84
|
-
return
|
|
120
|
+
return None
|
|
121
|
+
return None
|
|
122
|
+
|
|
123
|
+
def _is_workbook_open_in_excel(self, file_name: str) -> bool:
|
|
124
|
+
"""Comprueba si el libro ya está abierto en alguna sesión activa de Excel."""
|
|
125
|
+
return self._find_open_workbook_in_excel(file_name) is not None
|
|
85
126
|
|
|
86
127
|
def extract_by_cell_or_range(
|
|
87
128
|
self,
|
utils/excel_writer.py
CHANGED
|
@@ -24,17 +24,36 @@ except ImportError:
|
|
|
24
24
|
from .file_utils import resolve_file_path
|
|
25
25
|
|
|
26
26
|
|
|
27
|
-
def
|
|
28
|
-
"""
|
|
27
|
+
def _find_open_workbook_in_excel(target: str) -> Optional[Any]:
|
|
28
|
+
"""Busca y retorna el objeto Book de xlwings si está abierto en memoria."""
|
|
29
29
|
if not HAS_XLWINGS:
|
|
30
|
-
return
|
|
30
|
+
return None
|
|
31
31
|
try:
|
|
32
|
+
if not target or target.lower() in ["", "activo", "active", "libro_activo"]:
|
|
33
|
+
if len(xw.books) > 0:
|
|
34
|
+
return xw.books.active
|
|
35
|
+
return None
|
|
36
|
+
|
|
37
|
+
clean_target = os.path.basename(target).strip().lower()
|
|
38
|
+
name_no_ext = os.path.splitext(clean_target)[0]
|
|
39
|
+
|
|
32
40
|
for book in xw.books:
|
|
33
|
-
|
|
34
|
-
|
|
41
|
+
b_name = book.name.lower()
|
|
42
|
+
b_name_no_ext = os.path.splitext(b_name)[0]
|
|
43
|
+
b_fullname = getattr(book, 'fullname', '').lower()
|
|
44
|
+
|
|
45
|
+
if (clean_target == b_name or
|
|
46
|
+
name_no_ext == b_name_no_ext or
|
|
47
|
+
(b_fullname and clean_target == os.path.basename(b_fullname).lower())):
|
|
48
|
+
return book
|
|
35
49
|
except Exception:
|
|
36
|
-
return
|
|
37
|
-
return
|
|
50
|
+
return None
|
|
51
|
+
return None
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _is_workbook_open_in_excel(file_path: str, file_name: str) -> bool:
|
|
55
|
+
"""Comprueba si el libro ya está abierto en Excel activo."""
|
|
56
|
+
return _find_open_workbook_in_excel(file_name) is not None or _find_open_workbook_in_excel(file_path) is not None
|
|
38
57
|
|
|
39
58
|
|
|
40
59
|
def escribir_en_excel(
|
|
@@ -50,35 +69,45 @@ def escribir_en_excel(
|
|
|
50
69
|
) -> str:
|
|
51
70
|
"""
|
|
52
71
|
Escribe un DataFrame directamente en una celda específica de un libro de Excel.
|
|
53
|
-
|
|
54
|
-
Parámetros:
|
|
55
|
-
-----------
|
|
56
|
-
df : pd.DataFrame
|
|
57
|
-
El DataFrame con los datos a escribir.
|
|
58
|
-
archivo_excel : str
|
|
59
|
-
Ruta o nombre del archivo Excel.
|
|
60
|
-
hoja : str o int (por defecto 1)
|
|
61
|
-
Nombre o número de la hoja donde se escribirán los datos.
|
|
62
|
-
celda_inicio : str (por defecto 'A1')
|
|
63
|
-
Coordenada de la celda superior izquierda donde comenzará a escribirse (ej: 'B5', 'C4').
|
|
64
|
-
incluir_encabezados : bool (por defecto True)
|
|
65
|
-
Si escribe los nombres de las columnas en la primera fila.
|
|
66
|
-
incluir_indice : bool (por defecto False)
|
|
67
|
-
Si incluye la columna de índice de pandas.
|
|
68
|
-
guardar : bool (por defecto True)
|
|
69
|
-
Si guarda el archivo tras escribir.
|
|
70
|
-
archivo_abierto : bool o None
|
|
71
|
-
- True: Se conecta a la ventana activa de Excel.
|
|
72
|
-
- False: Trabaja en segundo plano cerrado.
|
|
73
|
-
- None: Detecta automáticamente si está abierto o cerrado.
|
|
74
|
-
crear_si_no_existe : bool (por defecto True)
|
|
75
|
-
Crea un nuevo archivo Excel si no existe en la ruta dada.
|
|
76
|
-
|
|
77
|
-
Retorna:
|
|
78
|
-
--------
|
|
79
|
-
str
|
|
80
|
-
Ruta absoluta del archivo modificado.
|
|
81
72
|
"""
|
|
73
|
+
# 1. Si está abierto en memoria o archivo_abierto=True, usar directamente xlwings
|
|
74
|
+
if HAS_XLWINGS and archivo_abierto is not False:
|
|
75
|
+
matched_book = _find_open_workbook_in_excel(archivo_excel)
|
|
76
|
+
if matched_book:
|
|
77
|
+
book = matched_book
|
|
78
|
+
# Seleccionar o crear hoja
|
|
79
|
+
sheet_names = [s.name for s in book.sheets]
|
|
80
|
+
if isinstance(hoja, int):
|
|
81
|
+
if 1 <= hoja <= len(book.sheets):
|
|
82
|
+
sht = book.sheets[hoja - 1]
|
|
83
|
+
else:
|
|
84
|
+
sht = book.sheets.add(f"Hoja{hoja}")
|
|
85
|
+
else:
|
|
86
|
+
if str(hoja) in sheet_names:
|
|
87
|
+
sht = book.sheets[str(hoja)]
|
|
88
|
+
else:
|
|
89
|
+
sht = book.sheets.add(str(hoja))
|
|
90
|
+
|
|
91
|
+
clean_cell = celda_inicio.replace("$", "").upper()
|
|
92
|
+
sht.range(clean_cell).options(
|
|
93
|
+
index=incluir_indice,
|
|
94
|
+
header=incluir_encabezados
|
|
95
|
+
).value = df
|
|
96
|
+
|
|
97
|
+
if guardar:
|
|
98
|
+
try: book.save()
|
|
99
|
+
except Exception: pass
|
|
100
|
+
|
|
101
|
+
return getattr(book, 'fullname', book.name)
|
|
102
|
+
|
|
103
|
+
if archivo_abierto is True and HAS_XLWINGS:
|
|
104
|
+
abiertos = [b.name for b in xw.books] if len(xw.books) > 0 else []
|
|
105
|
+
raise FileNotFoundError(
|
|
106
|
+
f"No se encontró ningún libro abierto en Excel con el nombre '{archivo_excel}'. "
|
|
107
|
+
f"Libros actualmente abiertos en Excel: {abiertos if abiertos else 'Ninguno (Excel no tiene libros abiertos)'}"
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
# 2. Si es archivo cerrado en disco
|
|
82
111
|
try:
|
|
83
112
|
resolved_path = resolve_file_path(archivo_excel)
|
|
84
113
|
except Exception:
|
|
@@ -87,7 +116,6 @@ def escribir_en_excel(
|
|
|
87
116
|
if not os.path.exists(resolved_path):
|
|
88
117
|
if crear_si_no_existe:
|
|
89
118
|
os.makedirs(os.path.dirname(resolved_path), exist_ok=True)
|
|
90
|
-
# Crear libro vacío con openpyxl o pandas
|
|
91
119
|
df_init = pd.DataFrame()
|
|
92
120
|
with pd.ExcelWriter(resolved_path, engine='openpyxl') as writer:
|
|
93
121
|
sheet_title = hoja if isinstance(hoja, str) else "Hoja1"
|
|
@@ -99,64 +127,45 @@ def escribir_en_excel(
|
|
|
99
127
|
|
|
100
128
|
# Intentar con xlwings si está disponible
|
|
101
129
|
if HAS_XLWINGS:
|
|
130
|
+
app = None
|
|
131
|
+
book = None
|
|
102
132
|
try:
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
try:
|
|
113
|
-
book = xw.books[file_name]
|
|
114
|
-
except Exception:
|
|
115
|
-
book = xw.Book(resolved_path)
|
|
133
|
+
app = xw.App(visible=False, add_book=False)
|
|
134
|
+
app.display_alerts = False
|
|
135
|
+
app.screen_updating = False
|
|
136
|
+
book = app.books.open(resolved_path)
|
|
137
|
+
|
|
138
|
+
sheet_names = [s.name for s in book.sheets]
|
|
139
|
+
if isinstance(hoja, int):
|
|
140
|
+
if 1 <= hoja <= len(book.sheets):
|
|
141
|
+
sht = book.sheets[hoja - 1]
|
|
116
142
|
else:
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
needs_close = True
|
|
122
|
-
|
|
123
|
-
# Seleccionar o crear hoja
|
|
124
|
-
sheet_names = [s.name for s in book.sheets]
|
|
125
|
-
if isinstance(hoja, int):
|
|
126
|
-
if 1 <= hoja <= len(book.sheets):
|
|
127
|
-
sht = book.sheets[hoja - 1]
|
|
128
|
-
else:
|
|
129
|
-
sht = book.sheets.add(f"Hoja{hoja}")
|
|
143
|
+
sht = book.sheets.add(f"Hoja{hoja}")
|
|
144
|
+
else:
|
|
145
|
+
if str(hoja) in sheet_names:
|
|
146
|
+
sht = book.sheets[str(hoja)]
|
|
130
147
|
else:
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
except Exception: pass
|
|
153
|
-
if app:
|
|
154
|
-
try: app.quit()
|
|
155
|
-
except Exception: pass
|
|
156
|
-
|
|
157
|
-
except Exception as e:
|
|
158
|
-
# Fallback a openpyxl si ocurre algún error COM
|
|
159
|
-
pass
|
|
148
|
+
sht = book.sheets.add(str(hoja))
|
|
149
|
+
|
|
150
|
+
clean_cell = celda_inicio.replace("$", "").upper()
|
|
151
|
+
sht.range(clean_cell).options(
|
|
152
|
+
index=incluir_indice,
|
|
153
|
+
header=incluir_encabezados
|
|
154
|
+
).value = df
|
|
155
|
+
|
|
156
|
+
if guardar:
|
|
157
|
+
book.save()
|
|
158
|
+
|
|
159
|
+
return resolved_path
|
|
160
|
+
finally:
|
|
161
|
+
if book:
|
|
162
|
+
try: book.close()
|
|
163
|
+
except Exception: pass
|
|
164
|
+
if app:
|
|
165
|
+
try: app.quit()
|
|
166
|
+
except Exception: pass
|
|
167
|
+
|
|
168
|
+
return resolved_path
|
|
160
169
|
|
|
161
170
|
# Fallback con openpyxl si xlwings no pudo ejecutarse
|
|
162
171
|
if HAS_OPENPYXL:
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|