tablas-python 0.1.2__py3-none-any.whl → 0.1.3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
helpers/table_manager.py CHANGED
@@ -58,14 +58,58 @@ class TableManager:
58
58
  auto_load : bool (por defecto True)
59
59
  Carga las tablas automáticamente al instanciar.
60
60
  """
61
+ self.archivo_abierto = archivo_abierto
62
+ self.preferir_xlwings = preferir_xlwings
63
+ self.tables: List[Union[RawTableInfo, RawExcelTableInfo, RawSQLiteTableInfo]] = []
64
+ self.is_open_in_memory = False
65
+
66
+ # 1. Si es Excel y archivo_abierto=True o None, buscar primero en libros abiertos en pantalla (xlwings)
67
+ if self.preferir_xlwings and archivo_abierto is not False:
68
+ try:
69
+ import xlwings as xw
70
+ if not file_path or file_path.lower() in ["", "activo", "active", "libro_activo"]:
71
+ matched_book = xw.books.active if len(xw.books) > 0 else None
72
+ else:
73
+ clean_target = os.path.basename(file_path).strip().lower()
74
+ name_no_ext = os.path.splitext(clean_target)[0]
75
+ matched_book = None
76
+ for b in xw.books:
77
+ b_name = b.name.lower()
78
+ b_name_no_ext = os.path.splitext(b_name)[0]
79
+ b_fullname = getattr(b, 'fullname', '').lower()
80
+ if (clean_target == b_name or
81
+ name_no_ext == b_name_no_ext or
82
+ (b_fullname and clean_target == os.path.basename(b_fullname).lower())):
83
+ matched_book = b
84
+ break
85
+
86
+ if matched_book:
87
+ self.file_path = getattr(matched_book, 'fullname', matched_book.name)
88
+ self.ext = os.path.splitext(matched_book.name)[1].lower() or ".xlsx"
89
+ self.is_open_in_memory = True
90
+ if auto_load:
91
+ self.cargar_tablas()
92
+ return
93
+ except Exception:
94
+ pass
95
+
96
+ if archivo_abierto is True:
97
+ try:
98
+ import xlwings as xw
99
+ abiertos = [b.name for b in xw.books] if len(xw.books) > 0 else []
100
+ except Exception:
101
+ abiertos = []
102
+ raise FileNotFoundError(
103
+ f"No se encontró ningún libro abierto en Excel con el nombre '{file_path}'. "
104
+ f"Libros actualmente abiertos en Excel: {abiertos if abiertos else 'Ninguno (Excel no tiene libros abiertos)'}"
105
+ )
106
+
107
+ # 2. Si no es un libro abierto en memoria, resolver la ruta física en disco
61
108
  self.file_path = resolve_file_path(file_path)
62
109
  if not os.path.exists(self.file_path):
63
110
  raise FileNotFoundError(f"No se encontró el archivo: {self.file_path}")
64
111
 
65
112
  self.ext = os.path.splitext(self.file_path)[1].lower()
66
- self.archivo_abierto = archivo_abierto
67
- self.preferir_xlwings = preferir_xlwings
68
- self.tables: List[Union[RawTableInfo, RawExcelTableInfo, RawSQLiteTableInfo]] = []
69
113
 
70
114
  valid_extensions = ['.pdf', '.xlsx', '.xls', '.xlsm', '.csv', '.db', '.sqlite', '.sqlite3', '.db3']
71
115
  if self.ext not in valid_extensions:
tablas_python/__init__.py CHANGED
@@ -32,7 +32,7 @@ from utils.data_helpers import (
32
32
  conciliar_tablas,
33
33
  )
34
34
 
35
- __version__ = "0.1.2"
35
+ __version__ = "0.1.3"
36
36
 
37
37
  __all__ = [
38
38
  "TableManager",
@@ -1,375 +1,375 @@
1
- Metadata-Version: 2.4
2
- Name: tablas-python
3
- Version: 0.1.2
4
- Summary: Suite modular para extraer, transformar, conciliar y exportar tablas desde PDF, Excel y SQLite a pandas.
5
- Author: Reciba
6
- License: MIT
7
- Project-URL: Homepage, https://github.com/Reciba/tablas_python
8
- Project-URL: Repository, https://github.com/Reciba/tablas_python.git
9
- Project-URL: Issues, https://github.com/Reciba/tablas_python/issues
10
- Keywords: tables,pdf,excel,xlwings,pdfplumber,sqlite,pandas,data-extraction,vlookup,conciliation
11
- Classifier: Programming Language :: Python :: 3
12
- Classifier: Programming Language :: Python :: 3.8
13
- Classifier: Programming Language :: Python :: 3.9
14
- Classifier: Programming Language :: Python :: 3.10
15
- Classifier: Programming Language :: Python :: 3.11
16
- Classifier: Programming Language :: Python :: 3.12
17
- Classifier: License :: OSI Approved :: MIT License
18
- Classifier: Operating System :: OS Independent
19
- Classifier: Topic :: Office/Business :: Financial :: Spreadsheet
20
- Classifier: Topic :: Scientific/Engineering :: Information Analysis
21
- Requires-Python: >=3.8
22
- Description-Content-Type: text/markdown
23
- License-File: LICENSE
24
- Requires-Dist: pandas>=1.5.0
25
- Requires-Dist: pdfplumber>=0.9.0
26
- Requires-Dist: openpyxl>=3.1.0
27
- Requires-Dist: xlwings>=0.30.0
28
- Requires-Dist: rich>=13.0.0
29
- Requires-Dist: numpy>=1.20.0
30
- Provides-Extra: dev
31
- Requires-Dist: build; extra == "dev"
32
- Requires-Dist: twine; extra == "dev"
33
- Requires-Dist: pytest; extra == "dev"
34
- Requires-Dist: reportlab>=3.6.0; extra == "dev"
35
- Dynamic: license-file
36
-
37
- # 📊 Tablas Python - Suite Integral de Procesamiento de Tablas
38
-
39
- Sistema modular en Python diseñado para **reconocer, extraer, transformar, conciliar, validar y exportar tablas** provenientes de múltiples fuentes:
40
- - 📄 **Documentos PDF** (usando `pdfplumber` con descarte de filas de basura iniciales)
41
- - 📊 **Archivos Excel / CSV** (usando `xlwings`, compatible con libros **abiertos** o **cerrados**, tablas oficiales `ListObjects`, celdas de inicio y rangos)
42
- - 🗄️ **Bases de Datos SQLite** (archivos `.db`, `.sqlite`, `.sqlite3`)
43
-
44
- Además, incluye un conjunto completo de herramientas para **cálculos rápidos (IVA, márgenes, % participación), formateo de monedas/porcentajes, cruces tipo `BUSCARV`, conciliación automática de tablas, escritura en vivo en Excel, unión masiva de carpetas y reportes de calidad de datos**.
45
-
46
- ---
47
-
48
- ## 📁 Arquitectura del Proyecto
49
-
50
- El código está estructurado de forma modular y desacoplada por capas:
51
-
52
- ```text
53
- tablas_python/
54
- │
55
- ├── main.py # Script principal con todas las utilidades importadas y listas para usar
56
- ├── requirements.txt # Dependencias del proyecto
57
- ├── README.md # Documentación técnica completa
58
- ├── .gitignore # Exclusiones de Git
59
- │
60
- ├── utils/ # Capa de bajo nivel (lógica desacoplada y reutilizable)
61
- │ ├── __init__.py # Exportaciones unificadas del paquete utils
62
- │ ├── file_utils.py # Resolución de rutas (relativas/absolutas) y nombres seguros
63
- │ ├── pdf_extractor.py # Extracción multipágina en PDF con pdfplumber
64
- │ ├── excel_extractor.py # Extracción en Excel con xlwings (tablas oficiales, celdas y rangos)
65
- │ ├── excel_writer.py # Escritura en Excel en vivo o en segundo plano con xlwings
66
- │ ├── sqlite_extractor.py # Extracción y consultas SQL en SQLite
67
- │ ├── table_cleaner.py # Limpieza, descarte de basura superior, tipos y normalización
68
- │ ├── batch_processor.py # Unión masiva de carpetas con múltiples archivos a 1 DataFrame
69
- │ ├── validator.py # Validación de esquemas, diagnóstico de calidad y duplicados
70
- │ ├── exporter.py # Exportación individual y por lotes a CSV/Excel (utf-8-sig / sep=';')
71
- │ └── data_helpers.py # Cálculos: BUSCARV, conciliación, IVA, %, totales, celdas y formato
72
- │
73
- ├── helpers/ # Capa de fachada y presentación
74
- │ ├── __init__.py # Exportaciones unificadas del paquete helpers
75
- │ ├── table_manager.py # TableManager unificado, obtener_tabla y exportar_archivo_a_csv
76
- │ └── display_helper.py # Visualizador enriquecido en consola con Rich
77
- │
78
- ├── samples/ # Archivos y generador de prueba
79
- │ ├── generate_samples.py # Script generador de datos de prueba
80
- │ ├── ejemplo_facturas.pdf # PDF de muestra con tablas y encabezados desplazados
81
- │ ├── ejemplo_inventario.xlsx # Excel con tablas oficiales, celdas específicas y múltiples hojas
82
- │ └── ejemplo_empresa.db # Base SQLite de muestra con tablas 'clientes', 'ventas' y vistas
83
- │
84
- └── exports/ # Carpeta por defecto para salidas CSV y Excel
85
- ```
86
-
87
- ---
88
-
89
- ## 🚀 Instalación
90
-
91
- Clona el repositorio e instala las dependencias:
92
-
93
- ```bash
94
- git clone https://github.com/Reciba/tablas_python.git
95
- cd tablas_python
96
- pip install -r requirements.txt
97
- ```
98
-
99
- ---
100
-
101
- ## 💡 Guía de Uso Completa
102
-
103
- ### 1. Extracción desde PDF (`pdfplumber`)
104
-
105
- Resuelve el problema donde las filas 1, 2 o 3 contienen títulos, membretes o metadatos irrelevantes y los encabezados reales comienzan más abajo (por ejemplo en la **fila 4**):
106
-
107
- ```python
108
- from helpers.table_manager import obtener_tabla, TableManager
109
-
110
- # Modo 1: En una sola línea indicando tabla 3 y fila 4 como encabezado:
111
- df = obtener_tabla("samples/ejemplo_facturas.pdf", tabla=3, fila_encabezado=4)
112
- print(df.head())
113
-
114
- # Modo 2: Con inspección previa usando TableManager:
115
- manager = TableManager("samples/ejemplo_facturas.pdf")
116
- manager.resumen() # Muestra cuántas tablas hay y en qué páginas
117
- manager.ver_crudo(tabla=1) # Muestra las primeras filas numeradas (Fila 1, Fila 2, Fila 3...)
118
- df_limpio = manager.get_df(tabla=1, fila_encabezado=4, skip_footer=1)
119
- ```
120
-
121
- ---
122
-
123
- ### 2. Extracción desde Excel con `xlwings`
124
-
125
- Soporta conectarse a archivos **abiertos en pantalla** (sin conflictos de bloqueo) o **cerrados en disco** (en segundo plano):
126
-
127
- ```python
128
- from helpers.table_manager import obtener_tabla
129
-
130
- # A) Por Nombre Oficial de Tabla de Excel (ListObject / Tabla con Formato):
131
- df_stock = obtener_tabla("samples/ejemplo_inventario.xlsx", tabla="TablaStock")
132
-
133
- # B) Por Celda de Inicio donde parte la tabla (ej. celda C4 en la hoja 'Despacho'):
134
- df_despacho = obtener_tabla(
135
- "samples/ejemplo_inventario.xlsx",
136
- celda_inicio="C4",
137
- hoja="Despacho"
138
- )
139
-
140
- # C) Por Rango Exacto:
141
- df_rango = obtener_tabla(
142
- "samples/ejemplo_inventario.xlsx",
143
- rango="C4:F8",
144
- hoja="Despacho"
145
- )
146
-
147
- # D) Control de archivo abierto/cerrado:
148
- # archivo_abierto=None (autodetecta), True (fuerza conexión al Excel abierto), False (abre oculto)
149
- df = obtener_tabla("reporte.xlsx", tabla=1, fila_encabezado=4, archivo_abierto=True)
150
- ```
151
-
152
- ---
153
-
154
- ### 3. Extracción y Consultas en SQLite (`.db`, `.sqlite`)
155
-
156
- ```python
157
- from helpers.table_manager import TableManager, obtener_tabla
158
-
159
- # A) Obtener tabla completa por nombre:
160
- df_clientes = obtener_tabla("samples/ejemplo_empresa.db", tabla="clientes")
161
-
162
- # B) Ejecutar consultas SQL personalizadas directamente a DataFrame:
163
- db = TableManager("samples/ejemplo_empresa.db")
164
- db.resumen() # Lista tablas y vistas
165
-
166
- df_ventas = db.query("""
167
- SELECT c.nombre AS cliente, c.ciudad, v.monto_neto, v.fecha
168
- FROM clientes c
169
- JOIN ventas v ON c.id_cliente = v.id_cliente
170
- WHERE v.estado = 'Pagado'
171
- """)
172
- ```
173
-
174
- ---
175
-
176
- ### 4. Búsqueda de Celdas por Nombre de Fila y Nombre de Columna
177
-
178
- Puedes acceder a cualquier celda específica mediante nombres sin depender de posiciones numéricas:
179
-
180
- ```python
181
- from utils.data_helpers import obtener_celda, modificar_celda
182
-
183
- # A) Usando la función helper obtener_celda:
184
- precio = obtener_celda(df, fila="PROD-101", columna="Precio Unitario")
185
- print("Precio PROD-101:", precio)
186
-
187
- # B) Usando set_index nativo de pandas con .at o .loc:
188
- df_idx = df.set_index("Codigo")
189
- precio = df_idx.at["PROD-101", "Precio Unitario"]
190
- cantidad = df_idx.at["PROD-101", "Cantidad"]
191
- total = float(precio) * float(cantidad)
192
-
193
- # C) Modificar una celda por su nombre:
194
- df_actualizado = modificar_celda(df, fila="PROD-101", columna="Precio Unitario", nuevo_valor=49.90)
195
- ```
196
-
197
- ---
198
-
199
- ### 5. Cruces de Datos con `buscar_v` (BUSCARV / VLOOKUP en 1 línea)
200
-
201
- Cruza dos DataFrames asociando datos a partir de una clave en común:
202
-
203
- ```python
204
- from utils.data_helpers import buscar_v
205
-
206
- # Trae el nombre del cliente desde df_clientes a df_ventas usando 'id_cliente':
207
- df_ventas["Nombre_Cliente"] = buscar_v(
208
- df_origen=df_ventas,
209
- df_destino=df_clientes,
210
- clave="id_cliente",
211
- columna_a_traer="nombre"
212
- )
213
- ```
214
-
215
- ---
216
-
217
- ### 6. Conciliación y Auditoría entre 2 Tablas (`conciliar_tablas`)
218
-
219
- Compara dos tablas (por ejemplo: extracto bancario vs. registro contable, o inventario físico vs. teórico):
220
-
221
- ```python
222
- from utils.data_helpers import conciliar_tablas
223
-
224
- resultado = conciliar_tablas(df_sistema, df_banco, clave="id_transaccion")
225
-
226
- print("Coincidentes:", resultado["coincidentes"]) # Idénticas en ambas tablas
227
- print("Diferencias:", resultado["diferencias"]) # Existen en ambas pero con valores distintos
228
- print("Solo en A:", resultado["solo_en_A"]) # Registros que faltan en B
229
- print("Solo en B:", resultado["solo_en_B"]) # Registros que faltan en A
230
- ```
231
-
232
- ---
233
-
234
- ### 7. Escritura en Vivo en Excel con `escribir_en_excel`
235
-
236
- Pega un DataFrame en una celda exacta de una plantilla Excel que tengas **abierta en pantalla** o **cerrada en disco**, conservando fórmulas y formatos:
237
-
238
- ```python
239
- from utils.excel_writer import escribir_en_excel
240
-
241
- escribir_en_excel(
242
- df=df_reporte,
243
- archivo_excel="plantilla_ventas.xlsx",
244
- hoja="Resumen",
245
- celda_inicio="B5",
246
- archivo_abierto=None # Detecta automáticamente si la ventana de Excel está abierta
247
- )
248
- ```
249
-
250
- ---
251
-
252
- ### 8. Consolidación Masiva de Carpetas (`unir_archivos_carpeta`)
253
-
254
- Lee y une decenas de archivos periódicos (facturas PDF, Excels mensuales, etc.) en un solo DataFrame maestro:
255
-
256
- ```python
257
- from utils.batch_processor import unir_archivos_carpeta
258
-
259
- df_consolidado = unir_archivos_carpeta(
260
- carpeta="facturas_2026/",
261
- extension="pdf",
262
- tabla=1,
263
- fila_encabezado=4
264
- )
265
- # Agrega automáticamente la columna 'Archivo_Origen' con el nombre de cada archivo
266
- ```
267
-
268
- ---
269
-
270
- ### 9. Calidad y Validación de Datos (`validar_dataframe` y `reporte_calidad`)
271
-
272
- ```python
273
- from utils.validator import validar_dataframe, reporte_calidad, detectar_duplicados
274
-
275
- # 1. Diagnóstico completo de nulos, completitud y tipos por columna:
276
- print(reporte_calidad(df))
277
-
278
- # 2. Validación de reglas de negocio antes de procesar:
279
- check = validar_dataframe(
280
- df,
281
- columnas_requeridas=["id_venta", "monto_neto", "fecha"],
282
- no_nulos=["id_venta", "monto_neto"],
283
- tipos_esperados={"monto_neto": "numeric"}
284
- )
285
-
286
- if not check["es_valido"]:
287
- print("❌ Errores encontrados:", check["errores"])
288
-
289
- # 3. Detección de registros duplicados:
290
- duplicados = detectar_duplicados(df, columnas_clave=["id_venta"])
291
- ```
292
-
293
- ---
294
-
295
- ### 10. Cálculos Rápidos y Formato para Reportes
296
-
297
- ```python
298
- from utils.data_helpers import (
299
- aplicar_impuesto,
300
- calcular_participacion,
301
- calcular_variacion,
302
- agregar_fila_totales,
303
- formatear_dataframe,
304
- formato_moneda,
305
- formato_porcentaje,
306
- limpiar_numero
307
- )
308
-
309
- # A) Calcular IVA (19%) y Total Bruto:
310
- df = aplicar_impuesto(df, col_neto="monto_neto", tasa=0.19)
311
-
312
- # B) Calcular % de participación sobre el total:
313
- df = calcular_participacion(df, columna_valor="Total Bruto")
314
-
315
- # C) Agregar fila final con totales:
316
- df = agregar_fila_totales(df, columnas_sumar=["monto_neto", "IVA (19%)", "Total Bruto"])
317
-
318
- # D) Formatear números a moneda ($) y porcentaje (%):
319
- df_formateado = formatear_dataframe(df, {
320
- "monto_neto": "moneda",
321
- "IVA (19%)": "moneda",
322
- "Total Bruto": "moneda",
323
- "% Participación": "porcentaje"
324
- })
325
- ```
326
-
327
- ---
328
-
329
- ### 11. Exportación Fácil a CSV y Excel
330
-
331
- ```python
332
- from utils.exporter import guardar_csv, guardar_excel
333
- from helpers.table_manager import exportar_archivo_a_csv
334
-
335
- # A) Guardar DataFrame individual a CSV (optimizado con ';' y utf-8-sig para Excel):
336
- guardar_csv(df, "exports/mi_tabla.csv")
337
-
338
- # B) Guardar a Excel:
339
- guardar_excel(df, "exports/mi_tabla.xlsx")
340
-
341
- # C) Exportar TODAS las tablas encontradas en un archivo a CSVs independientes:
342
- archivos = exportar_archivo_a_csv("samples/ejemplo_facturas.pdf", carpeta_salida="exports/pdf_tablas", fila_encabezado=4)
343
- ```
344
-
345
- ---
346
-
347
- ## 🖥️ Ejecución por Consola
348
-
349
- ```bash
350
- # 1. Ejecutar script principal de demostración:
351
- python main.py
352
-
353
- # 2. Modo interactivo en terminal (asistente guiado):
354
- python main.py -i
355
-
356
- # 3. Exportar todas las tablas de cualquier archivo por línea de comandos:
357
- python main.py -f "samples/ejemplo_facturas.pdf" -r 4 --export-all "exports/salida_csv"
358
- ```
359
-
360
- ---
361
-
362
- ## 📦 Tecnologías Utilizadas
363
-
364
- - **[pandas](https://pandas.pydata.org/)**: Manipulación y estructuras tabulares en DataFrames.
365
- - **[xlwings](https://docs.xlwings.org/)**: Integración avanzada con Microsoft Excel (archivos abiertos/cerrados, rangos y tablas).
366
- - **[pdfplumber](https://github.com/jsvine/pdfplumber)**: Extracción precisa de texto y tablas multipágina en PDFs.
367
- - **[openpyxl](https://openpyxl.readthedocs.io/)**: Soporte nativo y fallback para archivos `.xlsx`.
368
- - **[reportlab](https://www.reportlab.com/)**: Generación de PDFs de prueba.
369
- - **[rich](https://github.com/Textualize/rich)**: Visualización y formato de tablas en consola.
370
-
371
- ---
372
-
373
- ## 📄 Licencia
374
-
375
- Distribuido bajo licencia MIT. Consulta `LICENSE` para más información.
1
+ Metadata-Version: 2.4
2
+ Name: tablas-python
3
+ Version: 0.1.3
4
+ Summary: Suite modular para extraer, transformar, conciliar y exportar tablas desde PDF, Excel y SQLite a pandas.
5
+ Author: Reciba
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/Reciba/tablas_python
8
+ Project-URL: Repository, https://github.com/Reciba/tablas_python.git
9
+ Project-URL: Issues, https://github.com/Reciba/tablas_python/issues
10
+ Keywords: tables,pdf,excel,xlwings,pdfplumber,sqlite,pandas,data-extraction,vlookup,conciliation
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3.8
13
+ Classifier: Programming Language :: Python :: 3.9
14
+ Classifier: Programming Language :: Python :: 3.10
15
+ Classifier: Programming Language :: Python :: 3.11
16
+ Classifier: Programming Language :: Python :: 3.12
17
+ Classifier: License :: OSI Approved :: MIT License
18
+ Classifier: Operating System :: OS Independent
19
+ Classifier: Topic :: Office/Business :: Financial :: Spreadsheet
20
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
21
+ Requires-Python: >=3.8
22
+ Description-Content-Type: text/markdown
23
+ License-File: LICENSE
24
+ Requires-Dist: pandas>=1.5.0
25
+ Requires-Dist: pdfplumber>=0.9.0
26
+ Requires-Dist: openpyxl>=3.1.0
27
+ Requires-Dist: xlwings>=0.30.0
28
+ Requires-Dist: rich>=13.0.0
29
+ Requires-Dist: numpy>=1.20.0
30
+ Provides-Extra: dev
31
+ Requires-Dist: build; extra == "dev"
32
+ Requires-Dist: twine; extra == "dev"
33
+ Requires-Dist: pytest; extra == "dev"
34
+ Requires-Dist: reportlab>=3.6.0; extra == "dev"
35
+ Dynamic: license-file
36
+
37
+ # 📊 Tablas Python - Suite Integral de Procesamiento de Tablas
38
+
39
+ Sistema modular en Python diseñado para **reconocer, extraer, transformar, conciliar, validar y exportar tablas** provenientes de múltiples fuentes:
40
+ - 📄 **Documentos PDF** (usando `pdfplumber` con descarte de filas de basura iniciales)
41
+ - 📊 **Archivos Excel / CSV** (usando `xlwings`, compatible con libros **abiertos** o **cerrados**, tablas oficiales `ListObjects`, celdas de inicio y rangos)
42
+ - 🗄️ **Bases de Datos SQLite** (archivos `.db`, `.sqlite`, `.sqlite3`)
43
+
44
+ Además, incluye un conjunto completo de herramientas para **cálculos rápidos (IVA, márgenes, % participación), formateo de monedas/porcentajes, cruces tipo `BUSCARV`, conciliación automática de tablas, escritura en vivo en Excel, unión masiva de carpetas y reportes de calidad de datos**.
45
+
46
+ ---
47
+
48
+ ## 📁 Arquitectura del Proyecto
49
+
50
+ El código está estructurado de forma modular y desacoplada por capas:
51
+
52
+ ```text
53
+ tablas_python/
54
+ │
55
+ ├── main.py # Script principal con todas las utilidades importadas y listas para usar
56
+ ├── requirements.txt # Dependencias del proyecto
57
+ ├── README.md # Documentación técnica completa
58
+ ├── .gitignore # Exclusiones de Git
59
+ │
60
+ ├── utils/ # Capa de bajo nivel (lógica desacoplada y reutilizable)
61
+ │ ├── __init__.py # Exportaciones unificadas del paquete utils
62
+ │ ├── file_utils.py # Resolución de rutas (relativas/absolutas) y nombres seguros
63
+ │ ├── pdf_extractor.py # Extracción multipágina en PDF con pdfplumber
64
+ │ ├── excel_extractor.py # Extracción en Excel con xlwings (tablas oficiales, celdas y rangos)
65
+ │ ├── excel_writer.py # Escritura en Excel en vivo o en segundo plano con xlwings
66
+ │ ├── sqlite_extractor.py # Extracción y consultas SQL en SQLite
67
+ │ ├── table_cleaner.py # Limpieza, descarte de basura superior, tipos y normalización
68
+ │ ├── batch_processor.py # Unión masiva de carpetas con múltiples archivos a 1 DataFrame
69
+ │ ├── validator.py # Validación de esquemas, diagnóstico de calidad y duplicados
70
+ │ ├── exporter.py # Exportación individual y por lotes a CSV/Excel (utf-8-sig / sep=';')
71
+ │ └── data_helpers.py # Cálculos: BUSCARV, conciliación, IVA, %, totales, celdas y formato
72
+ │
73
+ ├── helpers/ # Capa de fachada y presentación
74
+ │ ├── __init__.py # Exportaciones unificadas del paquete helpers
75
+ │ ├── table_manager.py # TableManager unificado, obtener_tabla y exportar_archivo_a_csv
76
+ │ └── display_helper.py # Visualizador enriquecido en consola con Rich
77
+ │
78
+ ├── samples/ # Archivos y generador de prueba
79
+ │ ├── generate_samples.py # Script generador de datos de prueba
80
+ │ ├── ejemplo_facturas.pdf # PDF de muestra con tablas y encabezados desplazados
81
+ │ ├── ejemplo_inventario.xlsx # Excel con tablas oficiales, celdas específicas y múltiples hojas
82
+ │ └── ejemplo_empresa.db # Base SQLite de muestra con tablas 'clientes', 'ventas' y vistas
83
+ │
84
+ └── exports/ # Carpeta por defecto para salidas CSV y Excel
85
+ ```
86
+
87
+ ---
88
+
89
+ ## 🚀 Instalación
90
+
91
+ Clona el repositorio e instala las dependencias:
92
+
93
+ ```bash
94
+ git clone https://github.com/Reciba/tablas_python.git
95
+ cd tablas_python
96
+ pip install -r requirements.txt
97
+ ```
98
+
99
+ ---
100
+
101
+ ## 💡 Guía de Uso Completa
102
+
103
+ ### 1. Extracción desde PDF (`pdfplumber`)
104
+
105
+ Resuelve el problema donde las filas 1, 2 o 3 contienen títulos, membretes o metadatos irrelevantes y los encabezados reales comienzan más abajo (por ejemplo en la **fila 4**):
106
+
107
+ ```python
108
+ from helpers.table_manager import obtener_tabla, TableManager
109
+
110
+ # Modo 1: En una sola línea indicando tabla 3 y fila 4 como encabezado:
111
+ df = obtener_tabla("samples/ejemplo_facturas.pdf", tabla=3, fila_encabezado=4)
112
+ print(df.head())
113
+
114
+ # Modo 2: Con inspección previa usando TableManager:
115
+ manager = TableManager("samples/ejemplo_facturas.pdf")
116
+ manager.resumen() # Muestra cuántas tablas hay y en qué páginas
117
+ manager.ver_crudo(tabla=1) # Muestra las primeras filas numeradas (Fila 1, Fila 2, Fila 3...)
118
+ df_limpio = manager.get_df(tabla=1, fila_encabezado=4, skip_footer=1)
119
+ ```
120
+
121
+ ---
122
+
123
+ ### 2. Extracción desde Excel con `xlwings`
124
+
125
+ Soporta conectarse a archivos **abiertos en pantalla** (sin conflictos de bloqueo) o **cerrados en disco** (en segundo plano):
126
+
127
+ ```python
128
+ from helpers.table_manager import obtener_tabla
129
+
130
+ # A) Por Nombre Oficial de Tabla de Excel (ListObject / Tabla con Formato):
131
+ df_stock = obtener_tabla("samples/ejemplo_inventario.xlsx", tabla="TablaStock")
132
+
133
+ # B) Por Celda de Inicio donde parte la tabla (ej. celda C4 en la hoja 'Despacho'):
134
+ df_despacho = obtener_tabla(
135
+ "samples/ejemplo_inventario.xlsx",
136
+ celda_inicio="C4",
137
+ hoja="Despacho"
138
+ )
139
+
140
+ # C) Por Rango Exacto:
141
+ df_rango = obtener_tabla(
142
+ "samples/ejemplo_inventario.xlsx",
143
+ rango="C4:F8",
144
+ hoja="Despacho"
145
+ )
146
+
147
+ # D) Control de archivo abierto/cerrado:
148
+ # archivo_abierto=None (autodetecta), True (fuerza conexión al Excel abierto), False (abre oculto)
149
+ df = obtener_tabla("reporte.xlsx", tabla=1, fila_encabezado=4, archivo_abierto=True)
150
+ ```
151
+
152
+ ---
153
+
154
+ ### 3. Extracción y Consultas en SQLite (`.db`, `.sqlite`)
155
+
156
+ ```python
157
+ from helpers.table_manager import TableManager, obtener_tabla
158
+
159
+ # A) Obtener tabla completa por nombre:
160
+ df_clientes = obtener_tabla("samples/ejemplo_empresa.db", tabla="clientes")
161
+
162
+ # B) Ejecutar consultas SQL personalizadas directamente a DataFrame:
163
+ db = TableManager("samples/ejemplo_empresa.db")
164
+ db.resumen() # Lista tablas y vistas
165
+
166
+ df_ventas = db.query("""
167
+ SELECT c.nombre AS cliente, c.ciudad, v.monto_neto, v.fecha
168
+ FROM clientes c
169
+ JOIN ventas v ON c.id_cliente = v.id_cliente
170
+ WHERE v.estado = 'Pagado'
171
+ """)
172
+ ```
173
+
174
+ ---
175
+
176
+ ### 4. Búsqueda de Celdas por Nombre de Fila y Nombre de Columna
177
+
178
+ Puedes acceder a cualquier celda específica mediante nombres sin depender de posiciones numéricas:
179
+
180
+ ```python
181
+ from utils.data_helpers import obtener_celda, modificar_celda
182
+
183
+ # A) Usando la función helper obtener_celda:
184
+ precio = obtener_celda(df, fila="PROD-101", columna="Precio Unitario")
185
+ print("Precio PROD-101:", precio)
186
+
187
+ # B) Usando set_index nativo de pandas con .at o .loc:
188
+ df_idx = df.set_index("Codigo")
189
+ precio = df_idx.at["PROD-101", "Precio Unitario"]
190
+ cantidad = df_idx.at["PROD-101", "Cantidad"]
191
+ total = float(precio) * float(cantidad)
192
+
193
+ # C) Modificar una celda por su nombre:
194
+ df_actualizado = modificar_celda(df, fila="PROD-101", columna="Precio Unitario", nuevo_valor=49.90)
195
+ ```
196
+
197
+ ---
198
+
199
+ ### 5. Cruces de Datos con `buscar_v` (BUSCARV / VLOOKUP en 1 línea)
200
+
201
+ Cruza dos DataFrames asociando datos a partir de una clave en común:
202
+
203
+ ```python
204
+ from utils.data_helpers import buscar_v
205
+
206
+ # Trae el nombre del cliente desde df_clientes a df_ventas usando 'id_cliente':
207
+ df_ventas["Nombre_Cliente"] = buscar_v(
208
+ df_origen=df_ventas,
209
+ df_destino=df_clientes,
210
+ clave="id_cliente",
211
+ columna_a_traer="nombre"
212
+ )
213
+ ```
214
+
215
+ ---
216
+
217
+ ### 6. Conciliación y Auditoría entre 2 Tablas (`conciliar_tablas`)
218
+
219
+ Compara dos tablas (por ejemplo: extracto bancario vs. registro contable, o inventario físico vs. teórico):
220
+
221
+ ```python
222
+ from utils.data_helpers import conciliar_tablas
223
+
224
+ resultado = conciliar_tablas(df_sistema, df_banco, clave="id_transaccion")
225
+
226
+ print("Coincidentes:", resultado["coincidentes"]) # Idénticas en ambas tablas
227
+ print("Diferencias:", resultado["diferencias"]) # Existen en ambas pero con valores distintos
228
+ print("Solo en A:", resultado["solo_en_A"]) # Registros que faltan en B
229
+ print("Solo en B:", resultado["solo_en_B"]) # Registros que faltan en A
230
+ ```
231
+
232
+ ---
233
+
234
+ ### 7. Escritura en Vivo en Excel con `escribir_en_excel`
235
+
236
+ Pega un DataFrame en una celda exacta de una plantilla Excel que tengas **abierta en pantalla** o **cerrada en disco**, conservando fórmulas y formatos:
237
+
238
+ ```python
239
+ from utils.excel_writer import escribir_en_excel
240
+
241
+ escribir_en_excel(
242
+ df=df_reporte,
243
+ archivo_excel="plantilla_ventas.xlsx",
244
+ hoja="Resumen",
245
+ celda_inicio="B5",
246
+ archivo_abierto=None # Detecta automáticamente si la ventana de Excel está abierta
247
+ )
248
+ ```
249
+
250
+ ---
251
+
252
+ ### 8. Consolidación Masiva de Carpetas (`unir_archivos_carpeta`)
253
+
254
+ Lee y une decenas de archivos periódicos (facturas PDF, Excels mensuales, etc.) en un solo DataFrame maestro:
255
+
256
+ ```python
257
+ from utils.batch_processor import unir_archivos_carpeta
258
+
259
+ df_consolidado = unir_archivos_carpeta(
260
+ carpeta="facturas_2026/",
261
+ extension="pdf",
262
+ tabla=1,
263
+ fila_encabezado=4
264
+ )
265
+ # Agrega automáticamente la columna 'Archivo_Origen' con el nombre de cada archivo
266
+ ```
267
+
268
+ ---
269
+
270
+ ### 9. Calidad y Validación de Datos (`validar_dataframe` y `reporte_calidad`)
271
+
272
+ ```python
273
+ from utils.validator import validar_dataframe, reporte_calidad, detectar_duplicados
274
+
275
+ # 1. Diagnóstico completo de nulos, completitud y tipos por columna:
276
+ print(reporte_calidad(df))
277
+
278
+ # 2. Validación de reglas de negocio antes de procesar:
279
+ check = validar_dataframe(
280
+ df,
281
+ columnas_requeridas=["id_venta", "monto_neto", "fecha"],
282
+ no_nulos=["id_venta", "monto_neto"],
283
+ tipos_esperados={"monto_neto": "numeric"}
284
+ )
285
+
286
+ if not check["es_valido"]:
287
+ print("❌ Errores encontrados:", check["errores"])
288
+
289
+ # 3. Detección de registros duplicados:
290
+ duplicados = detectar_duplicados(df, columnas_clave=["id_venta"])
291
+ ```
292
+
293
+ ---
294
+
295
+ ### 10. Cálculos Rápidos y Formato para Reportes
296
+
297
+ ```python
298
+ from utils.data_helpers import (
299
+ aplicar_impuesto,
300
+ calcular_participacion,
301
+ calcular_variacion,
302
+ agregar_fila_totales,
303
+ formatear_dataframe,
304
+ formato_moneda,
305
+ formato_porcentaje,
306
+ limpiar_numero
307
+ )
308
+
309
+ # A) Calcular IVA (19%) y Total Bruto:
310
+ df = aplicar_impuesto(df, col_neto="monto_neto", tasa=0.19)
311
+
312
+ # B) Calcular % de participación sobre el total:
313
+ df = calcular_participacion(df, columna_valor="Total Bruto")
314
+
315
+ # C) Agregar fila final con totales:
316
+ df = agregar_fila_totales(df, columnas_sumar=["monto_neto", "IVA (19%)", "Total Bruto"])
317
+
318
+ # D) Formatear números a moneda ($) y porcentaje (%):
319
+ df_formateado = formatear_dataframe(df, {
320
+ "monto_neto": "moneda",
321
+ "IVA (19%)": "moneda",
322
+ "Total Bruto": "moneda",
323
+ "% Participación": "porcentaje"
324
+ })
325
+ ```
326
+
327
+ ---
328
+
329
+ ### 11. Exportación Fácil a CSV y Excel
330
+
331
+ ```python
332
+ from utils.exporter import guardar_csv, guardar_excel
333
+ from helpers.table_manager import exportar_archivo_a_csv
334
+
335
+ # A) Guardar DataFrame individual a CSV (optimizado con ';' y utf-8-sig para Excel):
336
+ guardar_csv(df, "exports/mi_tabla.csv")
337
+
338
+ # B) Guardar a Excel:
339
+ guardar_excel(df, "exports/mi_tabla.xlsx")
340
+
341
+ # C) Exportar TODAS las tablas encontradas en un archivo a CSVs independientes:
342
+ archivos = exportar_archivo_a_csv("samples/ejemplo_facturas.pdf", carpeta_salida="exports/pdf_tablas", fila_encabezado=4)
343
+ ```
344
+
345
+ ---
346
+
347
+ ## 🖥️ Ejecución por Consola
348
+
349
+ ```bash
350
+ # 1. Ejecutar script principal de demostración:
351
+ python main.py
352
+
353
+ # 2. Modo interactivo en terminal (asistente guiado):
354
+ python main.py -i
355
+
356
+ # 3. Exportar todas las tablas de cualquier archivo por línea de comandos:
357
+ python main.py -f "samples/ejemplo_facturas.pdf" -r 4 --export-all "exports/salida_csv"
358
+ ```
359
+
360
+ ---
361
+
362
+ ## 📦 Tecnologías Utilizadas
363
+
364
+ - **[pandas](https://pandas.pydata.org/)**: Manipulación y estructuras tabulares en DataFrames.
365
+ - **[xlwings](https://docs.xlwings.org/)**: Integración avanzada con Microsoft Excel (archivos abiertos/cerrados, rangos y tablas).
366
+ - **[pdfplumber](https://github.com/jsvine/pdfplumber)**: Extracción precisa de texto y tablas multipágina en PDFs.
367
+ - **[openpyxl](https://openpyxl.readthedocs.io/)**: Soporte nativo y fallback para archivos `.xlsx`.
368
+ - **[reportlab](https://www.reportlab.com/)**: Generación de PDFs de prueba.
369
+ - **[rich](https://github.com/Textualize/rich)**: Visualización y formato de tablas en consola.
370
+
371
+ ---
372
+
373
+ ## 📄 Licencia
374
+
375
+ Distribuido bajo licencia MIT. Consulta `LICENSE` para más información.
@@ -1,22 +1,22 @@
1
1
  helpers/__init__.py,sha256=hopndlywciR50FKi6W6Lo_cAOP_YyPsQV3xj8hfkH_Y,899
2
2
  helpers/display_helper.py,sha256=i0--Txkjc9__IIKwdkmpx98vzwM_Sekr0RzxvVGEPYA,7749
3
- helpers/table_manager.py,sha256=86MgHe-RtmZZYjkW_7lOQSy1cSGOBvleClEx_FhMg8g,16865
4
- tablas_python/__init__.py,sha256=QeyjiNr49IR0cgw_2OlW2mGzmsQqfqzzIoYfKyOonjM,1771
3
+ helpers/table_manager.py,sha256=85OZqMdDqgBBJaln2ehb3-t7RWS1Gowo6aud_zqVHAo,19063
4
+ tablas_python/__init__.py,sha256=Y-L9T9fn7gBiZdkXJ-qI9pe6bh7frMG71hgcqqcOqqM,1771
5
5
  tablas_python/cli.py,sha256=3BModRFfT8WFkbCSxM9cF_oBK87FetPMGm5F0GTZGNo,1957
6
- tablas_python-0.1.2.dist-info/licenses/LICENSE,sha256=DQi0EoD04d2_-pnEKrC4WmcmgR6O0WQcJdvma9ILoFQ,1063
6
+ tablas_python-0.1.3.dist-info/licenses/LICENSE,sha256=DQi0EoD04d2_-pnEKrC4WmcmgR6O0WQcJdvma9ILoFQ,1063
7
7
  utils/__init__.py,sha256=duEZSrdPZggUMdpJqJsNMQ5aLRaFcONBvhsdJuhCToo,2004
8
8
  utils/batch_processor.py,sha256=rE59qkb2sp5J0E5NU77U-CF82Aa-ukrjN2druwgS2Uk,4162
9
9
  utils/data_helpers.py,sha256=oJx5NPTpCNTHVvSCpqVeTnWGJsflYKl5v6lLUov08Hs,21758
10
- utils/excel_extractor.py,sha256=FqDqrBk-ls66-fk4R1bIblp0K532cIsC2PXgdcRX6a0,16191
11
- utils/excel_writer.py,sha256=MKbpAtuzxuD6wzHO6HRq7qxRKI5CpAJ-d8hN_GzSKns,6671
10
+ utils/excel_extractor.py,sha256=0WUhkV5eu9uUusU8zSHQB1ujFG7eQVqI-go4zqxmPFU,18364
11
+ utils/excel_writer.py,sha256=zH_Mtu3Cr_ND_wSHKFyvzH3j16gjVoANpM3dVFXKpr0,6991
12
12
  utils/exporter.py,sha256=Oo9Y6dXTg_7t6n824xoxv8d9yuhdHZNWfwLLXvSvtMw,4661
13
13
  utils/file_utils.py,sha256=Et3fEaRn70xxpIj5oz84N90ocACK2KkCUr1fO3h6Lgo,2302
14
14
  utils/pdf_extractor.py,sha256=3JRMqnEfhBVaUOeYfiNH1rReC4EoSbzDwk919x6647w,4768
15
15
  utils/sqlite_extractor.py,sha256=mvKlKY-nsd5cV73nn3xFoqg5dgagPJceRDJfPLG0cbs,5033
16
16
  utils/table_cleaner.py,sha256=ncyiunqpVdKParFcRKwQbCqyKZgpyKRN0aUM0Oug9wU,10465
17
17
  utils/validator.py,sha256=0UywK_D510ZQCcSkhFRKhEuw48788NM_RFVC-bIFZkE,5837
18
- tablas_python-0.1.2.dist-info/METADATA,sha256=1wrPGbeRuRclphZTIYvP_cl8zTwYs7Y5jPSUpc9rpH4,13986
19
- tablas_python-0.1.2.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
20
- tablas_python-0.1.2.dist-info/entry_points.txt,sha256=l29nooItcUpsFzN5LmxM1b-j0qih61vJB9tX8KUwlMg,89
21
- tablas_python-0.1.2.dist-info/top_level.txt,sha256=Foud-Tbf5pviUJmp7wc3v24IjIyypkvbcZt7HkrWvss,28
22
- tablas_python-0.1.2.dist-info/RECORD,,
18
+ tablas_python-0.1.3.dist-info/METADATA,sha256=rJMNr3NxGAcFwaz4xgAdmZ6IKJbCHPqiKXJGF4tG2MY,13611
19
+ tablas_python-0.1.3.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
20
+ tablas_python-0.1.3.dist-info/entry_points.txt,sha256=l29nooItcUpsFzN5LmxM1b-j0qih61vJB9tX8KUwlMg,89
21
+ tablas_python-0.1.3.dist-info/top_level.txt,sha256=Foud-Tbf5pviUJmp7wc3v24IjIyypkvbcZt7HkrWvss,28
22
+ tablas_python-0.1.3.dist-info/RECORD,,
utils/excel_extractor.py CHANGED
@@ -63,25 +63,66 @@ class ExcelTableExtractor:
63
63
  archivo_abierto: Optional[bool] = None,
64
64
  preferir_xlwings: bool = True
65
65
  ):
66
+ self.archivo_abierto = archivo_abierto
67
+ self.preferir_xlwings = preferir_xlwings and HAS_XLWINGS
68
+ self.book_name = os.path.basename(file_path) if file_path else ""
69
+ self.is_open_in_memory = False
70
+
71
+ # 1. Si el usuario indicó archivo_abierto=True o None, buscar primero en libros abiertos de xlwings
72
+ if self.preferir_xlwings and archivo_abierto is not False:
73
+ matched_book = self._find_open_workbook_in_excel(file_path)
74
+ if matched_book:
75
+ self.book_name = matched_book.name
76
+ self.file_path = getattr(matched_book, 'fullname', matched_book.name)
77
+ self.ext = os.path.splitext(matched_book.name)[1].lower() or ".xlsx"
78
+ self.is_open_in_memory = True
79
+ return
80
+
81
+ # Si forzó archivo_abierto=True pero no se encontró libro abierto en Excel
82
+ if archivo_abierto is True and self.preferir_xlwings:
83
+ abiertos = [b.name for b in xw.books] if HAS_XLWINGS and len(xw.books) > 0 else []
84
+ raise FileNotFoundError(
85
+ f"No se encontró ningún libro abierto en Excel con el nombre '{file_path}'. "
86
+ f"Libros actualmente abiertos en Excel: {abiertos if abiertos else 'Ninguno (Excel no tiene libros abiertos)'}"
87
+ )
88
+
89
+ # 2. Si es un archivo cerrado o en disco, resolver la ruta física
66
90
  self.file_path = resolve_file_path(file_path)
67
91
  if not os.path.exists(self.file_path):
68
92
  raise FileNotFoundError(f"No se encontró el archivo Excel en: {self.file_path}")
69
93
 
70
94
  self.ext = os.path.splitext(self.file_path)[1].lower()
71
- self.archivo_abierto = archivo_abierto
72
- self.preferir_xlwings = preferir_xlwings and HAS_XLWINGS
73
95
 
74
- def _is_workbook_open_in_excel(self, file_name: str) -> bool:
75
- """Comprueba si el libro ya está abierto en alguna sesión activa de Excel."""
96
+ def _find_open_workbook_in_excel(self, target: str) -> Optional[Any]:
97
+ """Busca y retorna el objeto Book de xlwings si está abierto en memoria."""
76
98
  if not HAS_XLWINGS:
77
- return False
99
+ return None
78
100
  try:
101
+ if not target or target.lower() in ["", "activo", "active", "libro_activo", "hoja_activa"]:
102
+ if len(xw.books) > 0:
103
+ return xw.books.active
104
+ return None
105
+
106
+ clean_target = os.path.basename(target).strip().lower()
107
+ name_no_ext = os.path.splitext(clean_target)[0]
108
+
79
109
  for book in xw.books:
80
- if book.name.lower() == file_name.lower() or book.fullname.lower() == self.file_path.lower():
81
- return True
110
+ b_name = book.name.lower()
111
+ b_name_no_ext = os.path.splitext(b_name)[0]
112
+ b_fullname = getattr(book, 'fullname', '').lower()
113
+
114
+ # Comparar con nombre completo, sin extensión o fullname
115
+ if (clean_target == b_name or
116
+ name_no_ext == b_name_no_ext or
117
+ (b_fullname and clean_target == os.path.basename(b_fullname).lower())):
118
+ return book
82
119
  except Exception:
83
- return False
84
- return False
120
+ return None
121
+ return None
122
+
123
+ def _is_workbook_open_in_excel(self, file_name: str) -> bool:
124
+ """Comprueba si el libro ya está abierto en alguna sesión activa de Excel."""
125
+ return self._find_open_workbook_in_excel(file_name) is not None
85
126
 
86
127
  def extract_by_cell_or_range(
87
128
  self,
utils/excel_writer.py CHANGED
@@ -24,17 +24,36 @@ except ImportError:
24
24
  from .file_utils import resolve_file_path
25
25
 
26
26
 
27
- def _is_workbook_open_in_excel(file_path: str, file_name: str) -> bool:
28
- """Comprueba si el libro ya está abierto en Excel activo."""
27
+ def _find_open_workbook_in_excel(target: str) -> Optional[Any]:
28
+ """Busca y retorna el objeto Book de xlwings si está abierto en memoria."""
29
29
  if not HAS_XLWINGS:
30
- return False
30
+ return None
31
31
  try:
32
+ if not target or target.lower() in ["", "activo", "active", "libro_activo"]:
33
+ if len(xw.books) > 0:
34
+ return xw.books.active
35
+ return None
36
+
37
+ clean_target = os.path.basename(target).strip().lower()
38
+ name_no_ext = os.path.splitext(clean_target)[0]
39
+
32
40
  for book in xw.books:
33
- if book.name.lower() == file_name.lower() or book.fullname.lower() == file_path.lower():
34
- return True
41
+ b_name = book.name.lower()
42
+ b_name_no_ext = os.path.splitext(b_name)[0]
43
+ b_fullname = getattr(book, 'fullname', '').lower()
44
+
45
+ if (clean_target == b_name or
46
+ name_no_ext == b_name_no_ext or
47
+ (b_fullname and clean_target == os.path.basename(b_fullname).lower())):
48
+ return book
35
49
  except Exception:
36
- return False
37
- return False
50
+ return None
51
+ return None
52
+
53
+
54
+ def _is_workbook_open_in_excel(file_path: str, file_name: str) -> bool:
55
+ """Comprueba si el libro ya está abierto en Excel activo."""
56
+ return _find_open_workbook_in_excel(file_name) is not None or _find_open_workbook_in_excel(file_path) is not None
38
57
 
39
58
 
40
59
  def escribir_en_excel(
@@ -50,35 +69,45 @@ def escribir_en_excel(
50
69
  ) -> str:
51
70
  """
52
71
  Escribe un DataFrame directamente en una celda específica de un libro de Excel.
53
-
54
- Parámetros:
55
- -----------
56
- df : pd.DataFrame
57
- El DataFrame con los datos a escribir.
58
- archivo_excel : str
59
- Ruta o nombre del archivo Excel.
60
- hoja : str o int (por defecto 1)
61
- Nombre o número de la hoja donde se escribirán los datos.
62
- celda_inicio : str (por defecto 'A1')
63
- Coordenada de la celda superior izquierda donde comenzará a escribirse (ej: 'B5', 'C4').
64
- incluir_encabezados : bool (por defecto True)
65
- Si escribe los nombres de las columnas en la primera fila.
66
- incluir_indice : bool (por defecto False)
67
- Si incluye la columna de índice de pandas.
68
- guardar : bool (por defecto True)
69
- Si guarda el archivo tras escribir.
70
- archivo_abierto : bool o None
71
- - True: Se conecta a la ventana activa de Excel.
72
- - False: Trabaja en segundo plano cerrado.
73
- - None: Detecta automáticamente si está abierto o cerrado.
74
- crear_si_no_existe : bool (por defecto True)
75
- Crea un nuevo archivo Excel si no existe en la ruta dada.
76
-
77
- Retorna:
78
- --------
79
- str
80
- Ruta absoluta del archivo modificado.
81
72
  """
73
+ # 1. Si está abierto en memoria o archivo_abierto=True, usar directamente xlwings
74
+ if HAS_XLWINGS and archivo_abierto is not False:
75
+ matched_book = _find_open_workbook_in_excel(archivo_excel)
76
+ if matched_book:
77
+ book = matched_book
78
+ # Seleccionar o crear hoja
79
+ sheet_names = [s.name for s in book.sheets]
80
+ if isinstance(hoja, int):
81
+ if 1 <= hoja <= len(book.sheets):
82
+ sht = book.sheets[hoja - 1]
83
+ else:
84
+ sht = book.sheets.add(f"Hoja{hoja}")
85
+ else:
86
+ if str(hoja) in sheet_names:
87
+ sht = book.sheets[str(hoja)]
88
+ else:
89
+ sht = book.sheets.add(str(hoja))
90
+
91
+ clean_cell = celda_inicio.replace("$", "").upper()
92
+ sht.range(clean_cell).options(
93
+ index=incluir_indice,
94
+ header=incluir_encabezados
95
+ ).value = df
96
+
97
+ if guardar:
98
+ try: book.save()
99
+ except Exception: pass
100
+
101
+ return getattr(book, 'fullname', book.name)
102
+
103
+ if archivo_abierto is True and HAS_XLWINGS:
104
+ abiertos = [b.name for b in xw.books] if len(xw.books) > 0 else []
105
+ raise FileNotFoundError(
106
+ f"No se encontró ningún libro abierto en Excel con el nombre '{archivo_excel}'. "
107
+ f"Libros actualmente abiertos en Excel: {abiertos if abiertos else 'Ninguno (Excel no tiene libros abiertos)'}"
108
+ )
109
+
110
+ # 2. Si es archivo cerrado en disco
82
111
  try:
83
112
  resolved_path = resolve_file_path(archivo_excel)
84
113
  except Exception:
@@ -87,7 +116,6 @@ def escribir_en_excel(
87
116
  if not os.path.exists(resolved_path):
88
117
  if crear_si_no_existe:
89
118
  os.makedirs(os.path.dirname(resolved_path), exist_ok=True)
90
- # Crear libro vacío con openpyxl o pandas
91
119
  df_init = pd.DataFrame()
92
120
  with pd.ExcelWriter(resolved_path, engine='openpyxl') as writer:
93
121
  sheet_title = hoja if isinstance(hoja, str) else "Hoja1"
@@ -99,64 +127,45 @@ def escribir_en_excel(
99
127
 
100
128
  # Intentar con xlwings si está disponible
101
129
  if HAS_XLWINGS:
130
+ app = None
131
+ book = None
102
132
  try:
103
- is_open = _is_workbook_open_in_excel(resolved_path, file_name)
104
- should_connect = archivo_abierto is True or (archivo_abierto is None and is_open)
105
-
106
- app = None
107
- book = None
108
- needs_close = False
109
-
110
- try:
111
- if should_connect:
112
- try:
113
- book = xw.books[file_name]
114
- except Exception:
115
- book = xw.Book(resolved_path)
133
+ app = xw.App(visible=False, add_book=False)
134
+ app.display_alerts = False
135
+ app.screen_updating = False
136
+ book = app.books.open(resolved_path)
137
+
138
+ sheet_names = [s.name for s in book.sheets]
139
+ if isinstance(hoja, int):
140
+ if 1 <= hoja <= len(book.sheets):
141
+ sht = book.sheets[hoja - 1]
116
142
  else:
117
- app = xw.App(visible=False, add_book=False)
118
- app.display_alerts = False
119
- app.screen_updating = False
120
- book = app.books.open(resolved_path)
121
- needs_close = True
122
-
123
- # Seleccionar o crear hoja
124
- sheet_names = [s.name for s in book.sheets]
125
- if isinstance(hoja, int):
126
- if 1 <= hoja <= len(book.sheets):
127
- sht = book.sheets[hoja - 1]
128
- else:
129
- sht = book.sheets.add(f"Hoja{hoja}")
143
+ sht = book.sheets.add(f"Hoja{hoja}")
144
+ else:
145
+ if str(hoja) in sheet_names:
146
+ sht = book.sheets[str(hoja)]
130
147
  else:
131
- if str(hoja) in sheet_names:
132
- sht = book.sheets[str(hoja)]
133
- else:
134
- sht = book.sheets.add(str(hoja))
135
-
136
- # Escribir DataFrame en la celda indicada
137
- clean_cell = celda_inicio.replace("$", "").upper()
138
- sht.range(clean_cell).options(
139
- index=incluir_indice,
140
- header=incluir_encabezados
141
- ).value = df
142
-
143
- if guardar:
144
- book.save()
145
-
146
- return resolved_path
147
-
148
- finally:
149
- if needs_close:
150
- if book:
151
- try: book.close()
152
- except Exception: pass
153
- if app:
154
- try: app.quit()
155
- except Exception: pass
156
-
157
- except Exception as e:
158
- # Fallback a openpyxl si ocurre algún error COM
159
- pass
148
+ sht = book.sheets.add(str(hoja))
149
+
150
+ clean_cell = celda_inicio.replace("$", "").upper()
151
+ sht.range(clean_cell).options(
152
+ index=incluir_indice,
153
+ header=incluir_encabezados
154
+ ).value = df
155
+
156
+ if guardar:
157
+ book.save()
158
+
159
+ return resolved_path
160
+ finally:
161
+ if book:
162
+ try: book.close()
163
+ except Exception: pass
164
+ if app:
165
+ try: app.quit()
166
+ except Exception: pass
167
+
168
+ return resolved_path
160
169
 
161
170
  # Fallback con openpyxl si xlwings no pudo ejecutarse
162
171
  if HAS_OPENPYXL: