exstruct 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
exstruct/core/cells.py ADDED
@@ -0,0 +1,950 @@
1
+ from __future__ import annotations
2
+
3
+ import logging
4
+ from collections import deque
5
+ from decimal import Decimal, InvalidOperation
6
+ import re
7
+ from pathlib import Path
8
+ from typing import Dict, List, Optional, Tuple
9
+
10
+ import numpy as np
11
+ import pandas as pd
12
+ import xlwings as xw
13
+ from openpyxl import load_workbook
14
+ from openpyxl.utils import get_column_letter, range_boundaries
15
+
16
+ from ..models import CellRow
17
+
18
+ logger = logging.getLogger(__name__)
19
+ _warned_keys: set[str] = set()
20
+
21
+ # Detection tuning parameters (can be overridden via set_table_detection_params)
22
+ _DETECTION_CONFIG = {
23
+ "table_score_threshold": 0.35,
24
+ "density_min": 0.05,
25
+ "coverage_min": 0.2,
26
+ "min_nonempty_cells": 3,
27
+ }
28
+
29
+
30
+ def warn_once(key: str, message: str) -> None:
31
+ if key not in _warned_keys:
32
+ logger.warning(message)
33
+ _warned_keys.add(key)
34
+
35
+
36
+ def extract_sheet_cells(file_path: Path) -> Dict[str, List[CellRow]]:
37
+ """Read all sheets via pandas and convert to CellRow list while skipping empty cells."""
38
+ dfs = pd.read_excel(file_path, header=None, sheet_name=None, dtype=str)
39
+ result: Dict[str, List[CellRow]] = {}
40
+ for sheet_name, df in dfs.items():
41
+ df = df.fillna("")
42
+ rows: List[CellRow] = []
43
+ for excel_row, row in enumerate(df.itertuples(index=False, name=None), start=1):
44
+ filtered: Dict[str, int | float | str] = {}
45
+ for j, v in enumerate(row):
46
+ s = "" if v is None else str(v)
47
+ if s.strip() == "":
48
+ continue
49
+ filtered[str(j)] = _coerce_numeric_preserve_format(s)
50
+ if not filtered:
51
+ continue
52
+ rows.append(CellRow(r=excel_row, c=filtered))
53
+ result[sheet_name] = rows
54
+ return result
55
+
56
+
57
+ def shrink_to_content(
58
+ sheet: xw.Sheet,
59
+ top: int,
60
+ left: int,
61
+ bottom: int,
62
+ right: int,
63
+ require_inside_border: bool = False,
64
+ min_nonempty_ratio: float = 0.0,
65
+ ) -> Tuple[int, int, int, int]:
66
+ """Trim a rectangle based on cell contents and optional border heuristics."""
67
+ rng = sheet.range((top, left), (bottom, right))
68
+ vals = rng.value
69
+ if vals is None:
70
+ vals = []
71
+ if not isinstance(vals, list):
72
+ vals = [[vals]]
73
+ elif vals and not isinstance(vals[0], list):
74
+ vals = [vals]
75
+ rows_n = len(vals)
76
+ cols_n = len(vals[0]) if rows_n else 0
77
+
78
+ def to_str(x):
79
+ return "" if x is None else str(x)
80
+
81
+ def is_empty_value(x):
82
+ return to_str(x).strip() == ""
83
+
84
+ def row_empty(i: int) -> bool:
85
+ return cols_n == 0 or all(is_empty_value(vals[i][j]) for j in range(cols_n))
86
+
87
+ def col_empty(j: int) -> bool:
88
+ return rows_n == 0 or all(is_empty_value(vals[i][j]) for i in range(rows_n))
89
+
90
+ def row_nonempty_ratio(i: int) -> float:
91
+ if cols_n == 0:
92
+ return 0.0
93
+ cnt = sum(1 for j in range(cols_n) if not is_empty_value(vals[i][j]))
94
+ return cnt / cols_n
95
+
96
+ def col_nonempty_ratio(j: int) -> float:
97
+ if rows_n == 0:
98
+ return 0.0
99
+ cnt = sum(1 for i in range(rows_n) if not is_empty_value(vals[i][j]))
100
+ return cnt / rows_n
101
+
102
+ XL_LINESTYLE_NONE = -4142
103
+ XL_INSIDE_VERTICAL = 11
104
+ XL_INSIDE_HORIZONTAL = 12
105
+
106
+ def column_has_inside_border(col_idx: int) -> bool:
107
+ if not require_inside_border:
108
+ return False
109
+ try:
110
+ for r in range(top, bottom + 1):
111
+ ls = (
112
+ sheet.api.Cells(r, left + col_idx)
113
+ .Borders(XL_INSIDE_VERTICAL)
114
+ .LineStyle
115
+ )
116
+ if ls is not None and ls != XL_LINESTYLE_NONE:
117
+ return True
118
+ except Exception:
119
+ pass
120
+ return False
121
+
122
+ def row_has_inside_border(row_idx: int) -> bool:
123
+ if not require_inside_border:
124
+ return False
125
+ try:
126
+ for c in range(left, right + 1):
127
+ ls = (
128
+ sheet.api.Cells(top + row_idx, c)
129
+ .Borders(XL_INSIDE_HORIZONTAL)
130
+ .LineStyle
131
+ )
132
+ if ls is not None and ls != XL_LINESTYLE_NONE:
133
+ return True
134
+ except Exception:
135
+ pass
136
+ return False
137
+
138
+ def should_trim_col(j: int) -> bool:
139
+ if col_empty(j):
140
+ return True
141
+ if require_inside_border and not column_has_inside_border(j):
142
+ return True
143
+ if min_nonempty_ratio > 0.0 and col_nonempty_ratio(j) < min_nonempty_ratio:
144
+ return True
145
+ return False
146
+
147
+ def should_trim_row(i: int) -> bool:
148
+ if row_empty(i):
149
+ return True
150
+ if require_inside_border and not row_has_inside_border(i):
151
+ return True
152
+ if min_nonempty_ratio > 0.0 and row_nonempty_ratio(i) < min_nonempty_ratio:
153
+ return True
154
+ return False
155
+
156
+ while left <= right and cols_n > 0:
157
+ if should_trim_col(0):
158
+ for i in range(rows_n):
159
+ if cols_n > 0:
160
+ vals[i].pop(0)
161
+ cols_n = len(vals[0]) if rows_n else 0
162
+ left += 1
163
+ else:
164
+ break
165
+ while top <= bottom and rows_n > 0:
166
+ if should_trim_row(0):
167
+ vals.pop(0)
168
+ rows_n = len(vals)
169
+ top += 1
170
+ else:
171
+ break
172
+ while left <= right and cols_n > 0:
173
+ if should_trim_col(cols_n - 1):
174
+ for i in range(rows_n):
175
+ if cols_n > 0:
176
+ vals[i].pop(cols_n - 1)
177
+ cols_n = len(vals[0]) if rows_n else 0
178
+ right -= 1
179
+ else:
180
+ break
181
+ while top <= bottom and rows_n > 0:
182
+ if should_trim_row(rows_n - 1):
183
+ vals.pop(rows_n - 1)
184
+ rows_n = len(vals)
185
+ bottom -= 1
186
+ else:
187
+ break
188
+ return top, left, bottom, right
189
+
190
+
191
+ def load_border_maps_xlsx(xlsx_path: Path, sheet_name: str):
192
+ wb = load_workbook(xlsx_path, data_only=True, read_only=False)
193
+ if sheet_name not in wb.sheetnames:
194
+ wb.close()
195
+ raise KeyError(f"Sheet '{sheet_name}' not found in {xlsx_path}")
196
+
197
+ ws = wb[sheet_name]
198
+ try:
199
+ min_col, min_row, max_col, max_row = range_boundaries(ws.calculate_dimension())
200
+ except Exception:
201
+ min_col, min_row, max_col, max_row = 1, 1, ws.max_column or 1, ws.max_row or 1
202
+
203
+ shape = (max_row + 1, max_col + 1)
204
+ has_border = np.zeros(shape, dtype=bool)
205
+ top_edge = np.zeros(shape, dtype=bool)
206
+ bottom_edge = np.zeros(shape, dtype=bool)
207
+ left_edge = np.zeros(shape, dtype=bool)
208
+ right_edge = np.zeros(shape, dtype=bool)
209
+
210
+ def edge_has_style(edge) -> bool:
211
+ if edge is None:
212
+ return False
213
+ style = getattr(edge, "style", None)
214
+ return style is not None and style != "none"
215
+
216
+ for r in range(min_row, max_row + 1):
217
+ for c in range(min_col, max_col + 1):
218
+ cell = ws.cell(row=r, column=c)
219
+ b = getattr(cell, "border", None)
220
+ if b is None:
221
+ continue
222
+
223
+ t = edge_has_style(b.top)
224
+ btm = edge_has_style(b.bottom)
225
+ l = edge_has_style(b.left)
226
+ rgt = edge_has_style(b.right)
227
+
228
+ if t or btm or l or rgt:
229
+ has_border[r, c] = True
230
+ if t:
231
+ top_edge[r, c] = True
232
+ if btm:
233
+ bottom_edge[r, c] = True
234
+ if l:
235
+ left_edge[r, c] = True
236
+ if rgt:
237
+ right_edge[r, c] = True
238
+
239
+ wb.close()
240
+ return has_border, top_edge, bottom_edge, left_edge, right_edge, max_row, max_col
241
+
242
+
243
+ def detect_border_clusters(has_border: np.ndarray, min_size: int = 4):
244
+ try:
245
+ from scipy.ndimage import label
246
+
247
+ structure = np.array([[0, 1, 0], [1, 1, 1], [0, 1, 0]], dtype=np.uint8)
248
+ lbl, num = label(has_border.astype(np.uint8), structure=structure) # type: ignore
249
+ rects: List[Tuple[int, int, int, int]] = []
250
+ for k in range(1, num + 1):
251
+ ys, xs = np.where(lbl == k)
252
+ if len(ys) < min_size:
253
+ continue
254
+ rects.append((int(ys.min()), int(xs.min()), int(ys.max()), int(xs.max())))
255
+ return rects
256
+ except Exception:
257
+ warn_once(
258
+ "scipy-missing",
259
+ "scipy is not available. Falling back to pure-Python BFS for connected components, which may be significantly slower.",
260
+ )
261
+ h, w = has_border.shape
262
+ visited = np.zeros_like(has_border, dtype=bool)
263
+ rects: List[Tuple[int, int, int, int]] = []
264
+ for r in range(h):
265
+ for c in range(w):
266
+ if not has_border[r, c] or visited[r, c]:
267
+ continue
268
+ q = deque([(r, c)])
269
+ visited[r, c] = True
270
+ ys = [r]
271
+ xs = [c]
272
+ while q:
273
+ yy, xx = q.popleft()
274
+ for dy, dx in ((1, 0), (-1, 0), (0, 1), (0, -1)):
275
+ ny, nx = yy + dy, xx + dx
276
+ if (
277
+ 0 <= ny < h
278
+ and 0 <= nx < w
279
+ and has_border[ny, nx]
280
+ and not visited[ny, nx]
281
+ ):
282
+ visited[ny, nx] = True
283
+ q.append((ny, nx))
284
+ ys.append(ny)
285
+ xs.append(nx)
286
+ if len(ys) >= min_size:
287
+ rects.append((min(ys), min(xs), max(ys), max(xs)))
288
+ return rects
289
+
290
+
291
+ def _get_values_block(ws, top, left, bottom, right):
292
+ vals = []
293
+ for row in ws.iter_rows(
294
+ min_row=top, max_row=bottom, min_col=left, max_col=right, values_only=True
295
+ ):
296
+ vals.append(list(row))
297
+ return vals
298
+
299
+
300
+ def _table_density_metrics(matrix) -> tuple[float, float]:
301
+ """
302
+ Given a 2D matrix (list of rows), return (density, coverage).
303
+ density: nonempty / total cells.
304
+ coverage: area of tight bounding box of nonempty cells divided by total area.
305
+ """
306
+ if not matrix:
307
+ return 0.0, 0.0
308
+ rows = len(matrix)
309
+ cols = len(matrix[0]) if rows else 0
310
+ if rows == 0 or cols == 0:
311
+ return 0.0, 0.0
312
+
313
+ nonempty_coords = []
314
+ for i, row in enumerate(matrix):
315
+ if not isinstance(row, list):
316
+ row = [row]
317
+ for j, v in enumerate(row):
318
+ if not (v is None or str(v).strip() == ""):
319
+ nonempty_coords.append((i, j))
320
+
321
+ total = rows * cols
322
+ if not nonempty_coords:
323
+ return 0.0, 0.0
324
+
325
+ nonempty = len(nonempty_coords)
326
+ density = nonempty / total
327
+
328
+ ys = [p[0] for p in nonempty_coords]
329
+ xs = [p[1] for p in nonempty_coords]
330
+ bbox_h = (max(ys) - min(ys) + 1)
331
+ bbox_w = (max(xs) - min(xs) + 1)
332
+ coverage = (bbox_h * bbox_w) / total if total > 0 else 0.0
333
+ return density, coverage
334
+
335
+
336
+ def _is_plausible_table(matrix) -> bool:
337
+ """
338
+ Heuristic: require at least 2 rows and 2 cols with meaningful data.
339
+ - At least 2 rows have 2 以上の非空セル
340
+ - At least 2 columns have 2 以上の非空セル
341
+ """
342
+ if not matrix:
343
+ return False
344
+ if not isinstance(matrix[0], list):
345
+ # normalize to 2D
346
+ matrix = [matrix]
347
+
348
+ rows = len(matrix)
349
+ cols = max((len(r) if isinstance(r, list) else 1) for r in matrix) if rows else 0
350
+ if rows < 2 or cols < 2:
351
+ return False
352
+
353
+ row_counts = []
354
+ col_counts = [0] * cols
355
+ for r in matrix:
356
+ if not isinstance(r, list):
357
+ r = [r]
358
+ cnt = 0
359
+ for j in range(cols):
360
+ v = r[j] if j < len(r) else None
361
+ if not (v is None or str(v).strip() == ""):
362
+ cnt += 1
363
+ col_counts[j] += 1
364
+ row_counts.append(cnt)
365
+
366
+ rows_with_two = sum(1 for c in row_counts if c >= 2)
367
+ cols_with_two = sum(1 for c in col_counts if c >= 2)
368
+ return rows_with_two >= 2 and cols_with_two >= 2
369
+
370
+
371
+ def _nonempty_clusters(matrix: List[List]) -> List[Tuple[int, int, int, int]]:
372
+ """Return bounding boxes of connected components of nonempty cells (4-neighbor)."""
373
+ if not matrix:
374
+ return []
375
+ rows = len(matrix)
376
+ cols = max(len(r) for r in matrix) if rows else 0
377
+ grid = [[False] * cols for _ in range(rows)]
378
+ for i, row in enumerate(matrix):
379
+ for j in range(cols):
380
+ v = row[j] if j < len(row) else None
381
+ if not (v is None or str(v).strip() == ""):
382
+ grid[i][j] = True
383
+ visited = [[False] * cols for _ in range(rows)]
384
+ boxes: List[Tuple[int, int, int, int]] = []
385
+
386
+ def bfs(sr: int, sc: int):
387
+ q = deque([(sr, sc)])
388
+ visited[sr][sc] = True
389
+ ys = [sr]
390
+ xs = [sc]
391
+ while q:
392
+ r, c = q.popleft()
393
+ for dr, dc in ((1, 0), (-1, 0), (0, 1), (0, -1)):
394
+ nr, nc = r + dr, c + dc
395
+ if 0 <= nr < rows and 0 <= nc < cols and grid[nr][nc] and not visited[nr][nc]:
396
+ visited[nr][nc] = True
397
+ q.append((nr, nc))
398
+ ys.append(nr)
399
+ xs.append(nc)
400
+ return min(ys), min(xs), max(ys), max(xs)
401
+
402
+ for i in range(rows):
403
+ for j in range(cols):
404
+ if grid[i][j] and not visited[i][j]:
405
+ boxes.append(bfs(i, j))
406
+ return boxes
407
+
408
+
409
+ def _normalize_matrix(matrix) -> List[List]:
410
+ if matrix is None:
411
+ return []
412
+ if not isinstance(matrix, list):
413
+ return [[matrix]]
414
+ if matrix and not isinstance(matrix[0], list):
415
+ return [matrix]
416
+ return matrix
417
+
418
+
419
+ def _header_like_row(row: List) -> bool:
420
+ nonempty = [v for v in row if not (v is None or str(v).strip() == "")]
421
+ if len(nonempty) < 2:
422
+ return False
423
+ str_like = 0
424
+ num_like = 0
425
+ for v in nonempty:
426
+ s = str(v)
427
+ if _INT_RE.match(s) or _FLOAT_RE.match(s):
428
+ num_like += 1
429
+ else:
430
+ str_like += 1
431
+ return str_like >= num_like and str_like >= 1
432
+
433
+
434
+ def _table_signal_score(matrix: List[List]) -> float:
435
+ density, coverage = _table_density_metrics(matrix)
436
+ header = any(_header_like_row(r) for r in matrix[:2]) # check first 2 rows
437
+
438
+ rows = len(matrix)
439
+ cols = max((len(r) if isinstance(r, list) else 1) for r in matrix) if rows else 0
440
+ row_counts = []
441
+ col_counts = [0] * cols if cols else []
442
+ for r in matrix:
443
+ if not isinstance(r, list):
444
+ r = [r]
445
+ cnt = 0
446
+ for j in range(cols):
447
+ v = r[j] if j < len(r) else None
448
+ if not (v is None or str(v).strip() == ""):
449
+ cnt += 1
450
+ if j < len(col_counts):
451
+ col_counts[j] += 1
452
+ row_counts.append(cnt)
453
+ rows_with_two = sum(1 for c in row_counts if c >= 2)
454
+ cols_with_two = sum(1 for c in col_counts if c >= 2)
455
+ structure_score = 0.1 if (rows_with_two >= 2 and cols_with_two >= 2) else 0.0
456
+
457
+ score = density
458
+ if header:
459
+ score += 0.2
460
+ if coverage > 0.5:
461
+ score += 0.1
462
+ score += structure_score
463
+ return score
464
+
465
+
466
+ def set_table_detection_params(
467
+ *,
468
+ table_score_threshold: float | None = None,
469
+ density_min: float | None = None,
470
+ coverage_min: float | None = None,
471
+ min_nonempty_cells: int | None = None,
472
+ ) -> None:
473
+ """
474
+ Configure table detection heuristics at runtime.
475
+ Any parameter left as None keeps its current value.
476
+ """
477
+ if table_score_threshold is not None:
478
+ _DETECTION_CONFIG["table_score_threshold"] = table_score_threshold
479
+ if density_min is not None:
480
+ _DETECTION_CONFIG["density_min"] = density_min
481
+ if coverage_min is not None:
482
+ _DETECTION_CONFIG["coverage_min"] = coverage_min
483
+ if min_nonempty_cells is not None:
484
+ _DETECTION_CONFIG["min_nonempty_cells"] = min_nonempty_cells
485
+
486
+
487
+ def shrink_to_content_openpyxl(
488
+ ws,
489
+ top: int,
490
+ left: int,
491
+ bottom: int,
492
+ right: int,
493
+ require_inside_border: bool,
494
+ top_edge,
495
+ bottom_edge,
496
+ left_edge,
497
+ right_edge,
498
+ min_nonempty_ratio: float = 0.0,
499
+ ) -> Tuple[int, int, int, int]:
500
+ vals = _get_values_block(ws, top, left, bottom, right)
501
+ rows_n = bottom - top + 1
502
+ cols_n = right - left + 1
503
+
504
+ def to_str(x):
505
+ return "" if x is None else str(x)
506
+
507
+ def is_empty_value(x):
508
+ return to_str(x).strip() == ""
509
+
510
+ def row_nonempty_ratio_local(i: int) -> float:
511
+ if cols_n <= 0:
512
+ return 0.0
513
+ row = vals[i]
514
+ cnt = sum(1 for v in row if not is_empty_value(v))
515
+ return cnt / cols_n
516
+
517
+ def col_nonempty_ratio_local(j: int) -> float:
518
+ if rows_n <= 0:
519
+ return 0.0
520
+ cnt = 0
521
+ for i in range(rows_n):
522
+ if not is_empty_value(vals[i][j]):
523
+ cnt += 1
524
+ return cnt / rows_n
525
+
526
+ def col_has_inside_border(j_abs: int) -> bool:
527
+ if not require_inside_border:
528
+ return False
529
+ count_pairs = 0
530
+ for r_abs in range(top, bottom + 1):
531
+ if (
532
+ j_abs > left
533
+ and right_edge[r_abs, j_abs - 1]
534
+ and left_edge[r_abs, j_abs]
535
+ ):
536
+ count_pairs += 1
537
+ return count_pairs > 0
538
+
539
+ def row_has_inside_border(i_abs: int) -> bool:
540
+ if not require_inside_border:
541
+ return False
542
+ count_pairs = 0
543
+ for c_abs in range(left, right + 1):
544
+ if i_abs > top and bottom_edge[i_abs - 1, c_abs] and top_edge[i_abs, c_abs]:
545
+ count_pairs += 1
546
+ return count_pairs > 0
547
+
548
+ while left <= right and cols_n > 0:
549
+ empty_col = all(
550
+ not (
551
+ top_edge[i, left]
552
+ or bottom_edge[i, left]
553
+ or left_edge[i, left]
554
+ or right_edge[i, left]
555
+ )
556
+ for i in range(top, bottom + 1)
557
+ )
558
+ if (
559
+ empty_col
560
+ or (require_inside_border and not col_has_inside_border(left))
561
+ or (
562
+ min_nonempty_ratio > 0.0
563
+ and col_nonempty_ratio_local(0) < min_nonempty_ratio
564
+ )
565
+ ):
566
+ for i in range(rows_n):
567
+ if cols_n > 0:
568
+ vals[i].pop(0)
569
+ cols_n -= 1
570
+ left += 1
571
+ else:
572
+ break
573
+ while top <= bottom and rows_n > 0:
574
+ empty_row = all(
575
+ not (
576
+ top_edge[top, j]
577
+ or bottom_edge[top, j]
578
+ or left_edge[top, j]
579
+ or right_edge[top, j]
580
+ )
581
+ for j in range(left, right + 1)
582
+ )
583
+ if (
584
+ empty_row
585
+ or (require_inside_border and not row_has_inside_border(top))
586
+ or (
587
+ min_nonempty_ratio > 0.0
588
+ and row_nonempty_ratio_local(0) < min_nonempty_ratio
589
+ )
590
+ ):
591
+ vals.pop(0)
592
+ rows_n -= 1
593
+ top += 1
594
+ else:
595
+ break
596
+ while left <= right and cols_n > 0:
597
+ empty_col = all(
598
+ not (
599
+ top_edge[i, right]
600
+ or bottom_edge[i, right]
601
+ or left_edge[i, right]
602
+ or right_edge[i, right]
603
+ )
604
+ for i in range(top, bottom + 1)
605
+ )
606
+ if (
607
+ empty_col
608
+ or (require_inside_border and not col_has_inside_border(right))
609
+ or (
610
+ min_nonempty_ratio > 0.0
611
+ and col_nonempty_ratio_local(cols_n - 1) < min_nonempty_ratio
612
+ )
613
+ ):
614
+ for i in range(rows_n):
615
+ if cols_n > 0:
616
+ vals[i].pop(cols_n - 1)
617
+ cols_n -= 1
618
+ right -= 1
619
+ else:
620
+ break
621
+ while top <= bottom and rows_n > 0:
622
+ empty_row = all(
623
+ not (
624
+ top_edge[bottom, j]
625
+ or bottom_edge[bottom, j]
626
+ or left_edge[bottom, j]
627
+ or right_edge[bottom, j]
628
+ )
629
+ for j in range(left, right + 1)
630
+ )
631
+ if (
632
+ empty_row
633
+ or (require_inside_border and not row_has_inside_border(bottom))
634
+ or (
635
+ min_nonempty_ratio > 0.0
636
+ and row_nonempty_ratio_local(rows_n - 1) < min_nonempty_ratio
637
+ )
638
+ ):
639
+ vals.pop(rows_n - 1)
640
+ rows_n -= 1
641
+ bottom -= 1
642
+ else:
643
+ break
644
+ return top, left, bottom, right
645
+
646
+
647
+ def detect_tables_xlwings(sheet: xw.Sheet) -> List[str]:
648
+ """Detect table-like ranges via COM: ListObjects first, then border clusters."""
649
+ tables: List[str] = []
650
+ try:
651
+ for lo in sheet.api.ListObjects:
652
+ rng = lo.Range
653
+ top_row = int(rng.Row)
654
+ left_col = int(rng.Column)
655
+ bottom_row = top_row + int(rng.Rows.Count) - 1
656
+ right_col = left_col + int(rng.Columns.Count) - 1
657
+ addr = rng.Address(RowAbsolute=False, ColumnAbsolute=False)
658
+ tables.append(addr)
659
+ except Exception:
660
+ pass
661
+
662
+ used = sheet.used_range
663
+ max_row = used.last_cell.row
664
+ max_col = used.last_cell.column
665
+ XL_EDGE_LEFT = 7
666
+ XL_EDGE_TOP = 8
667
+ XL_EDGE_BOTTOM = 9
668
+ XL_EDGE_RIGHT = 10
669
+ XL_INSIDE_VERTICAL = 11
670
+ XL_INSIDE_HORIZONTAL = 12
671
+ XL_LINESTYLE_NONE = -4142
672
+
673
+ def cell_has_any_border(r: int, c: int) -> bool:
674
+ try:
675
+ b = sheet.api.Cells(r, c).Borders
676
+ for idx in (
677
+ XL_EDGE_LEFT,
678
+ XL_EDGE_TOP,
679
+ XL_EDGE_RIGHT,
680
+ XL_EDGE_BOTTOM,
681
+ XL_INSIDE_VERTICAL,
682
+ XL_INSIDE_HORIZONTAL,
683
+ ):
684
+ ls = b(idx).LineStyle
685
+ if ls is not None and ls != XL_LINESTYLE_NONE:
686
+ try:
687
+ if getattr(b(idx), "Weight", 0) == 0:
688
+ continue
689
+ except Exception:
690
+ pass
691
+ return True
692
+ return False
693
+ except Exception:
694
+ return False
695
+
696
+ grid = [[False] * (max_col + 1) for _ in range(max_row + 1)]
697
+ for r in range(1, max_row + 1):
698
+ for c in range(1, max_col + 1):
699
+ if cell_has_any_border(r, c):
700
+ grid[r][c] = True
701
+ visited = [[False] * (max_col + 1) for _ in range(max_row + 1)]
702
+
703
+ def dfs(sr: int, sc: int, acc: List[Tuple[int, int]]):
704
+ stack = [(sr, sc)]
705
+ while stack:
706
+ rr, cc = stack.pop()
707
+ if not (1 <= rr <= max_row and 1 <= cc <= max_col):
708
+ continue
709
+ if visited[rr][cc] or not grid[rr][cc]:
710
+ continue
711
+ visited[rr][cc] = True
712
+ acc.append((rr, cc))
713
+ for dr, dc in ((1, 0), (-1, 0), (0, 1), (0, -1)):
714
+ stack.append((rr + dr, cc + dc))
715
+
716
+ clusters: List[Tuple[int, int, int, int]] = []
717
+ for r in range(1, max_row + 1):
718
+ for c in range(1, max_col + 1):
719
+ if grid[r][c] and not visited[r][c]:
720
+ cluster: List[Tuple[int, int]] = []
721
+ dfs(r, c, cluster)
722
+ if len(cluster) < 4:
723
+ continue
724
+ rows = [rc[0] for rc in cluster]
725
+ cols = [rc[1] for rc in cluster]
726
+ top_row = min(rows)
727
+ bottom_row = max(rows)
728
+ left_col = min(cols)
729
+ right_col = max(cols)
730
+ clusters.append((top_row, left_col, bottom_row, right_col))
731
+
732
+ def overlaps_for_merge(a, b):
733
+ # Do not merge if one rect fully contains the other (separate clusters like big frame vs small table)
734
+ contains = (
735
+ (a[0] <= b[0] and a[1] <= b[1] and a[2] >= b[2] and a[3] >= b[3])
736
+ or (b[0] <= a[0] and b[1] <= a[1] and b[2] >= a[2] and b[3] >= a[3])
737
+ )
738
+ if contains:
739
+ return False
740
+ return not (a[1] > b[3] or a[3] < b[1] or a[0] > b[2] or a[2] < b[0])
741
+
742
+ merged_rects: List[Tuple[int, int, int, int]] = []
743
+ for rect in sorted(clusters):
744
+ merged = False
745
+ for i, ex in enumerate(merged_rects):
746
+ if overlaps_for_merge(rect, ex):
747
+ merged_rects[i] = (
748
+ min(rect[0], ex[0]),
749
+ min(rect[1], ex[1]),
750
+ max(rect[2], ex[2]),
751
+ max(rect[3], ex[3]),
752
+ )
753
+ merged = True
754
+ break
755
+ if not merged:
756
+ merged_rects.append(rect)
757
+
758
+ dedup: set[str] = set()
759
+ for top_row, left_col, bottom_row, right_col in merged_rects:
760
+ top_row, left_col, bottom_row, right_col = shrink_to_content(
761
+ sheet, top_row, left_col, bottom_row, right_col, require_inside_border=False
762
+ )
763
+ try:
764
+ rng_vals = sheet.range((top_row, left_col), (bottom_row, right_col)).value
765
+ rng_vals = _normalize_matrix(rng_vals)
766
+ nonempty = sum(
767
+ 1
768
+ for row in rng_vals
769
+ for v in (row if isinstance(row, list) else [row])
770
+ if not (v is None or str(v).strip() == "")
771
+ )
772
+ except Exception:
773
+ nonempty = 0
774
+ if nonempty < _DETECTION_CONFIG["min_nonempty_cells"]:
775
+ continue
776
+ clusters = _nonempty_clusters(rng_vals)
777
+ for r0, c0, r1, c1 in clusters:
778
+ sub = [row[c0 : c1 + 1] for row in rng_vals[r0 : r1 + 1]]
779
+ density, coverage = _table_density_metrics(sub)
780
+ if density < _DETECTION_CONFIG["density_min"] and coverage < _DETECTION_CONFIG["coverage_min"]:
781
+ continue
782
+ if not _is_plausible_table(sub):
783
+ continue
784
+ score = _table_signal_score(sub)
785
+ if score < _DETECTION_CONFIG["table_score_threshold"]:
786
+ continue
787
+ addr = f"{xw.utils.col_name(left_col + c0)}{top_row + r0}:{xw.utils.col_name(left_col + c1)}{top_row + r1}"
788
+ if addr not in dedup:
789
+ dedup.add(addr)
790
+ tables.append(addr)
791
+ return tables
792
+
793
+
794
+ def detect_tables_openpyxl(xlsx_path: Path, sheet_name: str) -> List[str]:
795
+ wb = load_workbook(
796
+ xlsx_path,
797
+ data_only=True,
798
+ read_only=False,
799
+ )
800
+ ws = wb[sheet_name]
801
+ tables: List[str] = []
802
+ try:
803
+ openpyxl_tables = []
804
+ if hasattr(ws, "tables") and ws.tables:
805
+ if isinstance(ws.tables, dict):
806
+ openpyxl_tables = list(ws.tables.values())
807
+ else:
808
+ openpyxl_tables = list(ws.tables)
809
+ elif hasattr(ws, "_tables") and ws._tables: # type: ignore
810
+ openpyxl_tables = list(ws._tables) # type: ignore
811
+ for t in openpyxl_tables:
812
+ addr = t.ref
813
+ tables.append(addr)
814
+ except Exception:
815
+ pass
816
+
817
+ has_border, top_edge, bottom_edge, left_edge, right_edge, max_row, max_col = (
818
+ load_border_maps_xlsx(xlsx_path, sheet_name)
819
+ )
820
+ rects = detect_border_clusters(has_border, min_size=4)
821
+
822
+ def overlaps_for_merge(a, b):
823
+ contains = (
824
+ (a[0] <= b[0] and a[1] <= b[1] and a[2] >= b[2] and a[3] >= b[3])
825
+ or (b[0] <= a[0] and b[1] <= a[1] and b[2] >= a[2] and b[3] >= a[3])
826
+ )
827
+ if contains:
828
+ return False
829
+ return not (a[1] > b[3] or a[3] < b[1] or a[0] > b[2] or a[2] < b[0])
830
+
831
+ merged_rects: List[Tuple[int, int, int, int]] = []
832
+ for rect in sorted(rects):
833
+ merged = False
834
+ for i, ex in enumerate(merged_rects):
835
+ if overlaps_for_merge(rect, ex):
836
+ merged_rects[i] = (
837
+ min(rect[0], ex[0]),
838
+ min(rect[1], ex[1]),
839
+ max(rect[2], ex[2]),
840
+ max(rect[3], ex[3]),
841
+ )
842
+ merged = True
843
+ break
844
+ if not merged:
845
+ merged_rects.append(rect)
846
+
847
+ dedup: set[str] = set()
848
+ for top_row, left_col, bottom_row, right_col in merged_rects:
849
+ top_row, left_col, bottom_row, right_col = shrink_to_content_openpyxl(
850
+ ws,
851
+ top_row,
852
+ left_col,
853
+ bottom_row,
854
+ right_col,
855
+ require_inside_border=False,
856
+ top_edge=top_edge,
857
+ bottom_edge=bottom_edge,
858
+ left_edge=left_edge,
859
+ right_edge=right_edge,
860
+ min_nonempty_ratio=0.0,
861
+ )
862
+ vals_block = _get_values_block(ws, top_row, left_col, bottom_row, right_col)
863
+ vals_block = _normalize_matrix(vals_block)
864
+ nonempty = sum(
865
+ 1 for row in vals_block for v in row if not (v is None or str(v).strip() == "")
866
+ )
867
+ if nonempty < _DETECTION_CONFIG["min_nonempty_cells"]:
868
+ continue
869
+ clusters = _nonempty_clusters(vals_block)
870
+ for r0, c0, r1, c1 in clusters:
871
+ sub = [row[c0 : c1 + 1] for row in vals_block[r0 : r1 + 1]]
872
+ density, coverage = _table_density_metrics(sub)
873
+ if density < _DETECTION_CONFIG["density_min"] and coverage < _DETECTION_CONFIG["coverage_min"]:
874
+ continue
875
+ if not _is_plausible_table(sub):
876
+ continue
877
+ score = _table_signal_score(sub)
878
+ if score < _DETECTION_CONFIG["table_score_threshold"]:
879
+ continue
880
+ addr = f"{get_column_letter(left_col + c0)}{top_row + r0}:{get_column_letter(left_col + c1)}{top_row + r1}"
881
+ if addr not in dedup:
882
+ dedup.add(addr)
883
+ tables.append(addr)
884
+ wb.close()
885
+ return tables
886
+
887
+
888
+ def detect_tables(sheet: xw.Sheet) -> List[str]:
889
+ excel_path: Optional[Path] = None
890
+ try:
891
+ excel_path = Path(sheet.book.fullname)
892
+ except Exception:
893
+ excel_path = None
894
+
895
+ if excel_path and excel_path.suffix.lower() == ".xls":
896
+ warn_once(
897
+ f"xls-fallback::{excel_path}",
898
+ f"File '{excel_path.name}' is .xls (BIFF); openpyxl cannot read it. Falling back to COM-based detection (slower). Consider converting to .xlsx.",
899
+ )
900
+ return detect_tables_xlwings(sheet)
901
+
902
+ if excel_path and excel_path.suffix.lower() in (".xlsx", ".xlsm"):
903
+ try:
904
+ import openpyxl # noqa: F401
905
+ except Exception:
906
+ warn_once(
907
+ "openpyxl-missing",
908
+ "openpyxl is not installed. Falling back to COM-based detection (slower).",
909
+ )
910
+ return detect_tables_xlwings(sheet)
911
+
912
+ try:
913
+ return detect_tables_openpyxl(excel_path, sheet.name)
914
+ except Exception as e:
915
+ warn_once(
916
+ f"openpyxl-parse-fallback::{excel_path}::{sheet.name}",
917
+ f"openpyxl failed to parse '{excel_path.name}' (sheet '{sheet.name}'): {e!r}. Falling back to COM-based detection (slower).",
918
+ )
919
+ return detect_tables_xlwings(sheet)
920
+
921
+ warn_once(
922
+ "unknown-ext-fallback",
923
+ "Workbook path or extension is unavailable; falling back to COM-based detection (slower).",
924
+ )
925
+ return detect_tables_xlwings(sheet)
926
+
927
+
928
+ _INT_RE = re.compile(r"^[+-]?\d+$")
929
+ _FLOAT_RE = re.compile(r"^[+-]?\d*\.\d+$")
930
+
931
+
932
+ def _coerce_numeric_preserve_format(val: str) -> int | float | str:
933
+ """
934
+ Convert numeric-looking strings to int/float while keeping precision.
935
+ Integers stay int; decimals keep scale via Decimal before casting to float.
936
+ """
937
+ if _INT_RE.match(val):
938
+ try:
939
+ return int(val)
940
+ except Exception:
941
+ return val
942
+ if _FLOAT_RE.match(val):
943
+ try:
944
+ dec = Decimal(val)
945
+ scale = max(1, -dec.as_tuple().exponent)
946
+ quantized = dec.quantize(Decimal("1." + "0" * scale))
947
+ return float(quantized)
948
+ except (InvalidOperation, Exception):
949
+ return val
950
+ return val