langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,932 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from collections import Counter
|
|
5
|
+
from collections.abc import Iterable
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from openpyxl.formula import Tokenizer
|
|
10
|
+
from openpyxl.utils import column_index_from_string, get_column_letter, range_boundaries
|
|
11
|
+
from openpyxl.utils.cell import coordinate_to_tuple
|
|
12
|
+
|
|
13
|
+
from langparse.workbooks.labels import is_section_label, is_total_label
|
|
14
|
+
from langparse.workbooks.types import (
|
|
15
|
+
CandidateRegion,
|
|
16
|
+
CellSnapshot,
|
|
17
|
+
RegionAnchor,
|
|
18
|
+
SheetSnapshot,
|
|
19
|
+
SourceRef,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
_FORMULA_CELL_REF = re.compile(r"(?<![A-Z0-9_!])\$?([A-Z]{1,3})\$?([1-9][0-9]*)(?![A-Z0-9_(])")
|
|
23
|
+
_REASON_ORDER = {
|
|
24
|
+
"native_table_anchor": 0,
|
|
25
|
+
"defined_name_anchor": 1,
|
|
26
|
+
"print_area_anchor": 2,
|
|
27
|
+
"merged_title_anchor": 3,
|
|
28
|
+
"formula_continuity": 4,
|
|
29
|
+
"style_boundary": 5,
|
|
30
|
+
"density_boundary": 6,
|
|
31
|
+
"blank_band": 7,
|
|
32
|
+
"occupied_extent": 8,
|
|
33
|
+
}
|
|
34
|
+
_EXACT_ANCHOR_REASONS = {
|
|
35
|
+
"native_table_anchor",
|
|
36
|
+
"defined_name_anchor",
|
|
37
|
+
"print_area_anchor",
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass(frozen=True)
|
|
42
|
+
class _Rect:
|
|
43
|
+
min_column: int
|
|
44
|
+
min_row: int
|
|
45
|
+
max_column: int
|
|
46
|
+
max_row: int
|
|
47
|
+
|
|
48
|
+
@property
|
|
49
|
+
def area(self) -> int:
|
|
50
|
+
return (self.max_column - self.min_column + 1) * (self.max_row - self.min_row + 1)
|
|
51
|
+
|
|
52
|
+
@property
|
|
53
|
+
def range(self) -> str:
|
|
54
|
+
return (
|
|
55
|
+
f"{get_column_letter(self.min_column)}{self.min_row}:"
|
|
56
|
+
f"{get_column_letter(self.max_column)}{self.max_row}"
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@dataclass(frozen=True)
|
|
61
|
+
class _UsableAnchor:
|
|
62
|
+
anchor: RegionAnchor
|
|
63
|
+
rect: _Rect
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
@dataclass(frozen=True)
|
|
67
|
+
class _Cut:
|
|
68
|
+
orientation: str
|
|
69
|
+
boundary: int
|
|
70
|
+
reason: str
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def detect_candidate_regions(sheet: SheetSnapshot) -> list[CandidateRegion]:
|
|
74
|
+
"""Partition assignable cells using stable structural evidence.
|
|
75
|
+
|
|
76
|
+
The public interface intentionally remains a single deterministic function.
|
|
77
|
+
Native anchors, print areas, visual discontinuities, merged ranges and formula
|
|
78
|
+
references are implementation details hidden behind that seam.
|
|
79
|
+
"""
|
|
80
|
+
|
|
81
|
+
occupied = {coordinate: cell for coordinate, cell in sheet.cells.items() if _is_occupied(cell)}
|
|
82
|
+
if not occupied:
|
|
83
|
+
return []
|
|
84
|
+
|
|
85
|
+
positions = {coordinate: coordinate_to_tuple(coordinate) for coordinate in occupied}
|
|
86
|
+
coarse_regions = _initial_coarse_regions(sheet, positions)
|
|
87
|
+
blank_partitioned = len(coarse_regions) > 1
|
|
88
|
+
regions: list[CandidateRegion] = []
|
|
89
|
+
for coarse in coarse_regions:
|
|
90
|
+
regions.extend(
|
|
91
|
+
_partition_coarse_region(
|
|
92
|
+
sheet,
|
|
93
|
+
occupied,
|
|
94
|
+
positions,
|
|
95
|
+
coarse,
|
|
96
|
+
blank_partitioned=blank_partitioned,
|
|
97
|
+
)
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
regions.sort(
|
|
101
|
+
key=lambda region: (
|
|
102
|
+
coordinate_to_tuple(region.source_ref.range.split(":", 1)[0]),
|
|
103
|
+
region.source_ref.range,
|
|
104
|
+
)
|
|
105
|
+
)
|
|
106
|
+
assigned = [coordinate for region in regions for coordinate in region.cell_refs]
|
|
107
|
+
if Counter(assigned) != Counter(occupied.keys()):
|
|
108
|
+
raise RuntimeError("Candidate region partition violated cell ownership")
|
|
109
|
+
return regions
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _initial_coarse_regions(
|
|
113
|
+
sheet: SheetSnapshot,
|
|
114
|
+
positions: dict[str, tuple[int, int]],
|
|
115
|
+
) -> list[_Rect]:
|
|
116
|
+
coarse_regions = []
|
|
117
|
+
for min_row, max_row in _consecutive_groups(row for row, _ in positions.values()):
|
|
118
|
+
columns = {column for row, column in positions.values() if min_row <= row <= max_row}
|
|
119
|
+
band = _Rect(min(columns), min_row, max(columns), max_row)
|
|
120
|
+
if _shared_header_style(sheet, band):
|
|
121
|
+
# A styled header extending through blank input columns supplies
|
|
122
|
+
# positive continuity evidence; blank cells remain non-assignable.
|
|
123
|
+
coarse_regions.append(band)
|
|
124
|
+
else:
|
|
125
|
+
coarse_regions.extend(
|
|
126
|
+
_Rect(min_column, min_row, max_column, max_row)
|
|
127
|
+
for min_column, max_column in _consecutive_groups(columns)
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
coarse_regions = _join_section_bands(sheet, coarse_regions)
|
|
131
|
+
|
|
132
|
+
for anchor in sheet.region_anchors:
|
|
133
|
+
if anchor.kind != "excel_table" or anchor.source_ref.sheet_name != sheet.name:
|
|
134
|
+
continue
|
|
135
|
+
try:
|
|
136
|
+
anchor_rect = _rect_from_range(anchor.source_ref.range)
|
|
137
|
+
except ValueError:
|
|
138
|
+
continue
|
|
139
|
+
if not any(_contains(anchor_rect, *position) for position in positions.values()):
|
|
140
|
+
continue
|
|
141
|
+
overlapping = [rect for rect in coarse_regions if _rectangles_overlap(rect, anchor_rect)]
|
|
142
|
+
if not overlapping:
|
|
143
|
+
continue
|
|
144
|
+
coarse_regions = [
|
|
145
|
+
rect for rect in coarse_regions if not _rectangles_overlap(rect, anchor_rect)
|
|
146
|
+
]
|
|
147
|
+
coarse_regions.append(_bounding_rect([anchor_rect, *overlapping]))
|
|
148
|
+
coarse_regions = _merge_overlapping_rects(coarse_regions)
|
|
149
|
+
|
|
150
|
+
return sorted(
|
|
151
|
+
coarse_regions,
|
|
152
|
+
key=lambda rect: (rect.min_row, rect.min_column, rect.max_row, rect.max_column),
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _join_section_bands(sheet: SheetSnapshot, rectangles: list[_Rect]) -> list[_Rect]:
|
|
157
|
+
result: list[_Rect] = []
|
|
158
|
+
for rect in rectangles:
|
|
159
|
+
previous = result[-1] if result else None
|
|
160
|
+
first = sheet.cells.get(f"{get_column_letter(rect.min_column)}{rect.min_row}")
|
|
161
|
+
continuation = first is not None and (
|
|
162
|
+
(first.colspan > 1 and is_section_label(first.display_value))
|
|
163
|
+
or is_total_label(first.display_value)
|
|
164
|
+
)
|
|
165
|
+
if (
|
|
166
|
+
previous is not None
|
|
167
|
+
and continuation
|
|
168
|
+
and previous.min_column == rect.min_column
|
|
169
|
+
and rect.max_column <= previous.max_column
|
|
170
|
+
and 0 < rect.min_row - previous.max_row <= 2
|
|
171
|
+
and _shared_header_style(sheet, previous)
|
|
172
|
+
):
|
|
173
|
+
# Include the whole continuation row band, including disconnected
|
|
174
|
+
# amount cells following a merged subtotal label.
|
|
175
|
+
band_end = rect.max_row
|
|
176
|
+
result[-1] = _Rect(previous.min_column, previous.min_row, previous.max_column, band_end)
|
|
177
|
+
elif previous is not None and (
|
|
178
|
+
previous.min_row <= rect.min_row <= rect.max_row <= previous.max_row
|
|
179
|
+
and previous.min_column <= rect.min_column <= rect.max_column <= previous.max_column
|
|
180
|
+
):
|
|
181
|
+
continue
|
|
182
|
+
else:
|
|
183
|
+
result.append(rect)
|
|
184
|
+
return result
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _bounding_rect(rectangles: list[_Rect]) -> _Rect:
|
|
188
|
+
return _Rect(
|
|
189
|
+
min(rect.min_column for rect in rectangles),
|
|
190
|
+
min(rect.min_row for rect in rectangles),
|
|
191
|
+
max(rect.max_column for rect in rectangles),
|
|
192
|
+
max(rect.max_row for rect in rectangles),
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _merge_overlapping_rects(rectangles: list[_Rect]) -> list[_Rect]:
|
|
197
|
+
pending = list(rectangles)
|
|
198
|
+
merged: list[_Rect] = []
|
|
199
|
+
while pending:
|
|
200
|
+
current = pending.pop()
|
|
201
|
+
overlaps = [rect for rect in pending if _rectangles_overlap(current, rect)]
|
|
202
|
+
if overlaps:
|
|
203
|
+
pending = [rect for rect in pending if rect not in overlaps]
|
|
204
|
+
pending.append(_bounding_rect([current, *overlaps]))
|
|
205
|
+
else:
|
|
206
|
+
merged.append(current)
|
|
207
|
+
return merged
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def _partition_coarse_region(
|
|
211
|
+
sheet: SheetSnapshot,
|
|
212
|
+
occupied: dict[str, CellSnapshot],
|
|
213
|
+
positions: dict[str, tuple[int, int]],
|
|
214
|
+
coarse: _Rect,
|
|
215
|
+
*,
|
|
216
|
+
blank_partitioned: bool,
|
|
217
|
+
) -> list[CandidateRegion]:
|
|
218
|
+
selected_anchors, conflicts = _select_anchors(sheet, positions, coarse)
|
|
219
|
+
print_rects = _print_area_rects(sheet, coarse)
|
|
220
|
+
protected_rects = [selected.rect for selected in selected_anchors]
|
|
221
|
+
use_print_rects = len(print_rects) >= 2 and _pairwise_non_overlapping(print_rects)
|
|
222
|
+
if use_print_rects:
|
|
223
|
+
protected_rects.extend(print_rects)
|
|
224
|
+
return _partition_rect(
|
|
225
|
+
sheet,
|
|
226
|
+
occupied,
|
|
227
|
+
positions,
|
|
228
|
+
coarse,
|
|
229
|
+
selected_anchors,
|
|
230
|
+
conflicts,
|
|
231
|
+
print_rects,
|
|
232
|
+
use_print_rects=use_print_rects,
|
|
233
|
+
protected_rects=protected_rects,
|
|
234
|
+
partition_reasons=frozenset(),
|
|
235
|
+
blank_partitioned=blank_partitioned,
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _partition_rect(
|
|
240
|
+
sheet: SheetSnapshot,
|
|
241
|
+
occupied: dict[str, CellSnapshot],
|
|
242
|
+
positions: dict[str, tuple[int, int]],
|
|
243
|
+
rect: _Rect,
|
|
244
|
+
selected_anchors: list[_UsableAnchor],
|
|
245
|
+
conflicts: list[tuple[_UsableAnchor, _UsableAnchor]],
|
|
246
|
+
print_rects: list[_Rect],
|
|
247
|
+
*,
|
|
248
|
+
use_print_rects: bool,
|
|
249
|
+
protected_rects: list[_Rect],
|
|
250
|
+
partition_reasons: frozenset[str],
|
|
251
|
+
blank_partitioned: bool,
|
|
252
|
+
) -> list[CandidateRegion]:
|
|
253
|
+
cell_refs = sorted(
|
|
254
|
+
(coordinate for coordinate, position in positions.items() if _contains(rect, *position)),
|
|
255
|
+
key=coordinate_to_tuple,
|
|
256
|
+
)
|
|
257
|
+
if not cell_refs:
|
|
258
|
+
return []
|
|
259
|
+
|
|
260
|
+
cuts = _candidate_cuts(
|
|
261
|
+
sheet,
|
|
262
|
+
occupied,
|
|
263
|
+
positions,
|
|
264
|
+
rect,
|
|
265
|
+
selected_anchors,
|
|
266
|
+
print_rects,
|
|
267
|
+
use_print_rects=use_print_rects,
|
|
268
|
+
protected_rects=protected_rects,
|
|
269
|
+
)
|
|
270
|
+
if cuts:
|
|
271
|
+
cut = cuts[0]
|
|
272
|
+
first, second = _split_rect(rect, cut)
|
|
273
|
+
inherited_reasons = partition_reasons
|
|
274
|
+
if cut.reason not in _EXACT_ANCHOR_REASONS:
|
|
275
|
+
inherited_reasons = frozenset((*partition_reasons, cut.reason))
|
|
276
|
+
return [
|
|
277
|
+
*_partition_rect(
|
|
278
|
+
sheet,
|
|
279
|
+
occupied,
|
|
280
|
+
positions,
|
|
281
|
+
first,
|
|
282
|
+
selected_anchors,
|
|
283
|
+
conflicts,
|
|
284
|
+
print_rects,
|
|
285
|
+
use_print_rects=use_print_rects,
|
|
286
|
+
protected_rects=protected_rects,
|
|
287
|
+
partition_reasons=inherited_reasons,
|
|
288
|
+
blank_partitioned=blank_partitioned,
|
|
289
|
+
),
|
|
290
|
+
*_partition_rect(
|
|
291
|
+
sheet,
|
|
292
|
+
occupied,
|
|
293
|
+
positions,
|
|
294
|
+
second,
|
|
295
|
+
selected_anchors,
|
|
296
|
+
conflicts,
|
|
297
|
+
print_rects,
|
|
298
|
+
use_print_rects=use_print_rects,
|
|
299
|
+
protected_rects=protected_rects,
|
|
300
|
+
partition_reasons=inherited_reasons,
|
|
301
|
+
blank_partitioned=blank_partitioned,
|
|
302
|
+
),
|
|
303
|
+
]
|
|
304
|
+
|
|
305
|
+
source_rect = _anchored_source_rect(rect, cell_refs, positions, selected_anchors)
|
|
306
|
+
reasons = _region_reasons(
|
|
307
|
+
sheet,
|
|
308
|
+
occupied,
|
|
309
|
+
positions,
|
|
310
|
+
source_rect,
|
|
311
|
+
selected_anchors,
|
|
312
|
+
print_rects,
|
|
313
|
+
partition_reasons=partition_reasons,
|
|
314
|
+
blank_partitioned=blank_partitioned,
|
|
315
|
+
)
|
|
316
|
+
diagnostics = [
|
|
317
|
+
{
|
|
318
|
+
"reason_code": "overlapping_native_anchors",
|
|
319
|
+
"kept_kind": kept.anchor.kind,
|
|
320
|
+
"kept_range": kept.rect.range,
|
|
321
|
+
"kept_name": kept.anchor.name,
|
|
322
|
+
"kept_scope": kept.anchor.scope,
|
|
323
|
+
"rejected_kind": rejected.anchor.kind,
|
|
324
|
+
"rejected_range": rejected.rect.range,
|
|
325
|
+
"rejected_name": rejected.anchor.name,
|
|
326
|
+
"rejected_scope": rejected.anchor.scope,
|
|
327
|
+
}
|
|
328
|
+
for kept, rejected in conflicts
|
|
329
|
+
if _rectangles_overlap(source_rect, kept.rect)
|
|
330
|
+
or _rectangles_overlap(source_rect, rejected.rect)
|
|
331
|
+
]
|
|
332
|
+
return [
|
|
333
|
+
CandidateRegion(
|
|
334
|
+
source_ref=SourceRef(sheet_name=sheet.name, range=source_rect.range),
|
|
335
|
+
cell_refs=cell_refs,
|
|
336
|
+
confidence=_region_confidence(reasons, diagnostics),
|
|
337
|
+
features={
|
|
338
|
+
"row_count": source_rect.max_row - source_rect.min_row + 1,
|
|
339
|
+
"column_count": source_rect.max_column - source_rect.min_column + 1,
|
|
340
|
+
"occupied_count": len(cell_refs),
|
|
341
|
+
"density": len(cell_refs) / source_rect.area,
|
|
342
|
+
},
|
|
343
|
+
diagnostics=diagnostics,
|
|
344
|
+
reason_codes=reasons,
|
|
345
|
+
)
|
|
346
|
+
]
|
|
347
|
+
|
|
348
|
+
|
|
349
|
+
def _candidate_cuts(
|
|
350
|
+
sheet: SheetSnapshot,
|
|
351
|
+
occupied: dict[str, CellSnapshot],
|
|
352
|
+
positions: dict[str, tuple[int, int]],
|
|
353
|
+
rect: _Rect,
|
|
354
|
+
selected_anchors: list[_UsableAnchor],
|
|
355
|
+
print_rects: list[_Rect],
|
|
356
|
+
*,
|
|
357
|
+
use_print_rects: bool,
|
|
358
|
+
protected_rects: list[_Rect],
|
|
359
|
+
) -> list[_Cut]:
|
|
360
|
+
cuts: dict[tuple[str, int], _Cut] = {}
|
|
361
|
+
# Formula-free exports are common. Never scan every cell for every possible
|
|
362
|
+
# row cut; tokenize each formula once and index the boundaries it protects.
|
|
363
|
+
formula_columns, formula_rows = _formula_boundaries(occupied, positions, rect)
|
|
364
|
+
evidence = [
|
|
365
|
+
(selected.rect, _anchor_reason(selected.anchor.kind))
|
|
366
|
+
for selected in selected_anchors
|
|
367
|
+
if _rectangles_overlap(selected.rect, rect)
|
|
368
|
+
]
|
|
369
|
+
if use_print_rects:
|
|
370
|
+
evidence.extend(
|
|
371
|
+
(print_rect, "print_area_anchor")
|
|
372
|
+
for print_rect in print_rects
|
|
373
|
+
if _rectangles_overlap(print_rect, rect)
|
|
374
|
+
)
|
|
375
|
+
for evidence_rect, reason in evidence:
|
|
376
|
+
for cut in _rect_edge_cuts(evidence_rect, rect, reason):
|
|
377
|
+
if _cut_is_safe(sheet, rect, cut, protected_rects):
|
|
378
|
+
_offer_cut(cuts, cut)
|
|
379
|
+
|
|
380
|
+
for boundary in range(rect.min_column, rect.max_column):
|
|
381
|
+
key = ("vertical", boundary)
|
|
382
|
+
if key in cuts:
|
|
383
|
+
continue
|
|
384
|
+
cut = _Cut("vertical", boundary, "style_boundary")
|
|
385
|
+
if not _cut_is_safe(sheet, rect, cut, protected_rects):
|
|
386
|
+
continue
|
|
387
|
+
if boundary in formula_columns:
|
|
388
|
+
continue
|
|
389
|
+
if _is_style_boundary(sheet, rect, boundary):
|
|
390
|
+
cuts[key] = cut
|
|
391
|
+
elif _is_density_boundary(occupied, positions, rect, boundary):
|
|
392
|
+
cuts[key] = _Cut("vertical", boundary, "density_boundary")
|
|
393
|
+
|
|
394
|
+
for boundary in range(rect.min_row, rect.max_row):
|
|
395
|
+
key = ("horizontal", boundary)
|
|
396
|
+
if key in cuts:
|
|
397
|
+
continue
|
|
398
|
+
cut = _Cut("horizontal", boundary, "style_boundary")
|
|
399
|
+
if not _cut_is_safe(sheet, rect, cut, protected_rects):
|
|
400
|
+
continue
|
|
401
|
+
if boundary in formula_rows:
|
|
402
|
+
continue
|
|
403
|
+
if _is_row_style_boundary(sheet, rect, boundary):
|
|
404
|
+
cuts[key] = cut
|
|
405
|
+
elif _is_row_density_boundary(occupied, positions, rect, boundary):
|
|
406
|
+
cuts[key] = _Cut("horizontal", boundary, "density_boundary")
|
|
407
|
+
|
|
408
|
+
return sorted(
|
|
409
|
+
cuts.values(),
|
|
410
|
+
key=lambda cut: (
|
|
411
|
+
_REASON_ORDER.get(cut.reason, 99),
|
|
412
|
+
0 if cut.orientation == "vertical" else 1,
|
|
413
|
+
cut.boundary,
|
|
414
|
+
),
|
|
415
|
+
)
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def _rect_edge_cuts(evidence: _Rect, rect: _Rect, reason: str) -> list[_Cut]:
|
|
419
|
+
cuts = []
|
|
420
|
+
if rect.min_column < evidence.min_column <= rect.max_column:
|
|
421
|
+
cuts.append(_Cut("vertical", evidence.min_column - 1, reason))
|
|
422
|
+
if rect.min_column <= evidence.max_column < rect.max_column:
|
|
423
|
+
cuts.append(_Cut("vertical", evidence.max_column, reason))
|
|
424
|
+
if rect.min_row < evidence.min_row <= rect.max_row:
|
|
425
|
+
cuts.append(_Cut("horizontal", evidence.min_row - 1, reason))
|
|
426
|
+
if rect.min_row <= evidence.max_row < rect.max_row:
|
|
427
|
+
cuts.append(_Cut("horizontal", evidence.max_row, reason))
|
|
428
|
+
return cuts
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def _offer_cut(cuts: dict[tuple[str, int], _Cut], candidate: _Cut) -> None:
|
|
432
|
+
key = (candidate.orientation, candidate.boundary)
|
|
433
|
+
current = cuts.get(key)
|
|
434
|
+
if current is None or _REASON_ORDER.get(candidate.reason, 99) < _REASON_ORDER.get(
|
|
435
|
+
current.reason,
|
|
436
|
+
99,
|
|
437
|
+
):
|
|
438
|
+
cuts[key] = candidate
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
def _cut_is_safe(
|
|
442
|
+
sheet: SheetSnapshot,
|
|
443
|
+
rect: _Rect,
|
|
444
|
+
cut: _Cut,
|
|
445
|
+
protected_rects: list[_Rect],
|
|
446
|
+
) -> bool:
|
|
447
|
+
if cut.orientation == "vertical":
|
|
448
|
+
if _merged_range_crosses_column(sheet, rect, cut.boundary):
|
|
449
|
+
return False
|
|
450
|
+
return not any(
|
|
451
|
+
_rectangles_overlap(protected, rect)
|
|
452
|
+
and protected.min_column <= cut.boundary < protected.max_column
|
|
453
|
+
for protected in protected_rects
|
|
454
|
+
)
|
|
455
|
+
if _merged_range_crosses_row(sheet, rect, cut.boundary):
|
|
456
|
+
return False
|
|
457
|
+
return not any(
|
|
458
|
+
_rectangles_overlap(protected, rect)
|
|
459
|
+
and protected.min_row <= cut.boundary < protected.max_row
|
|
460
|
+
for protected in protected_rects
|
|
461
|
+
)
|
|
462
|
+
|
|
463
|
+
|
|
464
|
+
def _split_rect(rect: _Rect, cut: _Cut) -> tuple[_Rect, _Rect]:
|
|
465
|
+
if cut.orientation == "vertical":
|
|
466
|
+
return (
|
|
467
|
+
_Rect(rect.min_column, rect.min_row, cut.boundary, rect.max_row),
|
|
468
|
+
_Rect(cut.boundary + 1, rect.min_row, rect.max_column, rect.max_row),
|
|
469
|
+
)
|
|
470
|
+
return (
|
|
471
|
+
_Rect(rect.min_column, rect.min_row, rect.max_column, cut.boundary),
|
|
472
|
+
_Rect(rect.min_column, cut.boundary + 1, rect.max_column, rect.max_row),
|
|
473
|
+
)
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
def _select_anchors(
|
|
477
|
+
sheet: SheetSnapshot,
|
|
478
|
+
positions: dict[str, tuple[int, int]],
|
|
479
|
+
coarse: _Rect,
|
|
480
|
+
) -> tuple[list[_UsableAnchor], list[tuple[_UsableAnchor, _UsableAnchor]]]:
|
|
481
|
+
usable = []
|
|
482
|
+
for anchor in sheet.region_anchors:
|
|
483
|
+
if anchor.source_ref.sheet_name != sheet.name:
|
|
484
|
+
continue
|
|
485
|
+
try:
|
|
486
|
+
rect = _rect_from_range(anchor.source_ref.range)
|
|
487
|
+
except ValueError:
|
|
488
|
+
continue
|
|
489
|
+
if not _rectangles_overlap(rect, coarse) or not _rect_inside(rect, coarse):
|
|
490
|
+
continue
|
|
491
|
+
count = sum(_contains(rect, row, column) for row, column in positions.values())
|
|
492
|
+
if not count:
|
|
493
|
+
continue
|
|
494
|
+
if anchor.kind == "defined_name" and (
|
|
495
|
+
rect.max_column == rect.min_column
|
|
496
|
+
or rect.max_row == rect.min_row
|
|
497
|
+
or count / rect.area < 0.5
|
|
498
|
+
):
|
|
499
|
+
continue
|
|
500
|
+
if anchor.kind not in {"excel_table", "defined_name"}:
|
|
501
|
+
continue
|
|
502
|
+
usable.append(_UsableAnchor(anchor, rect))
|
|
503
|
+
|
|
504
|
+
usable.sort(
|
|
505
|
+
key=lambda item: (
|
|
506
|
+
{"excel_table": 0, "defined_name": 1}.get(item.anchor.kind, 99),
|
|
507
|
+
-item.rect.area,
|
|
508
|
+
item.rect.min_row,
|
|
509
|
+
item.rect.min_column,
|
|
510
|
+
item.anchor.name or "",
|
|
511
|
+
)
|
|
512
|
+
)
|
|
513
|
+
selected: list[_UsableAnchor] = []
|
|
514
|
+
conflicts: list[tuple[_UsableAnchor, _UsableAnchor]] = []
|
|
515
|
+
for candidate in usable:
|
|
516
|
+
conflicting = next(
|
|
517
|
+
(item for item in selected if _rectangles_overlap(item.rect, candidate.rect)),
|
|
518
|
+
None,
|
|
519
|
+
)
|
|
520
|
+
if conflicting is not None:
|
|
521
|
+
if conflicting.rect == candidate.rect:
|
|
522
|
+
continue
|
|
523
|
+
conflicts.append((conflicting, candidate))
|
|
524
|
+
continue
|
|
525
|
+
selected.append(candidate)
|
|
526
|
+
return selected, conflicts
|
|
527
|
+
|
|
528
|
+
|
|
529
|
+
def _print_area_rects(sheet: SheetSnapshot, coarse: _Rect) -> list[_Rect]:
|
|
530
|
+
rectangles = []
|
|
531
|
+
for value in sheet.print_area:
|
|
532
|
+
local_range = value.rsplit("!", 1)[-1].replace("$", "")
|
|
533
|
+
try:
|
|
534
|
+
rect = _rect_from_range(local_range)
|
|
535
|
+
except ValueError:
|
|
536
|
+
continue
|
|
537
|
+
if _rect_inside(rect, coarse):
|
|
538
|
+
rectangles.append(rect)
|
|
539
|
+
return sorted(
|
|
540
|
+
set(rectangles),
|
|
541
|
+
key=lambda item: (item.min_row, item.min_column, item.max_row, item.max_column),
|
|
542
|
+
)
|
|
543
|
+
|
|
544
|
+
|
|
545
|
+
def _anchored_source_rect(
|
|
546
|
+
segment_rect: _Rect,
|
|
547
|
+
cell_refs: list[str],
|
|
548
|
+
positions: dict[str, tuple[int, int]],
|
|
549
|
+
anchors: list[_UsableAnchor],
|
|
550
|
+
) -> _Rect:
|
|
551
|
+
for selected in anchors:
|
|
552
|
+
anchor_cells = {
|
|
553
|
+
coordinate
|
|
554
|
+
for coordinate, position in positions.items()
|
|
555
|
+
if _contains(selected.rect, *position)
|
|
556
|
+
}
|
|
557
|
+
if anchor_cells == set(cell_refs) and _rect_inside(selected.rect, segment_rect):
|
|
558
|
+
return selected.rect
|
|
559
|
+
return segment_rect
|
|
560
|
+
|
|
561
|
+
|
|
562
|
+
def _region_reasons(
|
|
563
|
+
sheet: SheetSnapshot,
|
|
564
|
+
occupied: dict[str, CellSnapshot],
|
|
565
|
+
positions: dict[str, tuple[int, int]],
|
|
566
|
+
rect: _Rect,
|
|
567
|
+
anchors: list[_UsableAnchor],
|
|
568
|
+
print_rects: list[_Rect],
|
|
569
|
+
*,
|
|
570
|
+
partition_reasons: frozenset[str],
|
|
571
|
+
blank_partitioned: bool,
|
|
572
|
+
) -> list[str]:
|
|
573
|
+
reasons = set(partition_reasons)
|
|
574
|
+
for selected in anchors:
|
|
575
|
+
if selected.rect == rect:
|
|
576
|
+
reasons.add(_anchor_reason(selected.anchor.kind))
|
|
577
|
+
if rect in print_rects:
|
|
578
|
+
reasons.add("print_area_anchor")
|
|
579
|
+
if any(
|
|
580
|
+
cell.formula is not None and _local_formula_refs(cell.formula)
|
|
581
|
+
for coordinate, cell in occupied.items()
|
|
582
|
+
if _contains(rect, *positions[coordinate])
|
|
583
|
+
):
|
|
584
|
+
reasons.add("formula_continuity")
|
|
585
|
+
if any(_rectangles_overlap(rect, merged) for merged in _merged_rects(sheet)):
|
|
586
|
+
reasons.add("merged_title_anchor")
|
|
587
|
+
if blank_partitioned:
|
|
588
|
+
reasons.add("blank_band")
|
|
589
|
+
if not reasons or reasons == {"formula_continuity"}:
|
|
590
|
+
reasons.add("occupied_extent")
|
|
591
|
+
return sorted(reasons, key=lambda item: (_REASON_ORDER.get(item, 99), item))
|
|
592
|
+
|
|
593
|
+
|
|
594
|
+
def _region_confidence(reasons: list[str], diagnostics: list[dict[str, Any]]) -> float:
|
|
595
|
+
if "native_table_anchor" in reasons:
|
|
596
|
+
confidence = 0.98
|
|
597
|
+
elif "print_area_anchor" in reasons or "defined_name_anchor" in reasons:
|
|
598
|
+
confidence = 0.95
|
|
599
|
+
elif "style_boundary" in reasons:
|
|
600
|
+
confidence = 0.9
|
|
601
|
+
elif "density_boundary" in reasons:
|
|
602
|
+
confidence = 0.82
|
|
603
|
+
else:
|
|
604
|
+
confidence = 1.0
|
|
605
|
+
if diagnostics:
|
|
606
|
+
confidence = min(confidence, 0.75)
|
|
607
|
+
return confidence
|
|
608
|
+
|
|
609
|
+
|
|
610
|
+
def _is_style_boundary(sheet: SheetSnapshot, rect: _Rect, boundary: int) -> bool:
|
|
611
|
+
if boundary - rect.min_column + 1 < 2 or rect.max_column - boundary < 2:
|
|
612
|
+
return False
|
|
613
|
+
if _shared_header_style(sheet, rect):
|
|
614
|
+
return False
|
|
615
|
+
comparisons = []
|
|
616
|
+
for row in range(rect.min_row, rect.max_row + 1):
|
|
617
|
+
left = sheet.cells.get(f"{get_column_letter(boundary)}{row}")
|
|
618
|
+
right = sheet.cells.get(f"{get_column_letter(boundary + 1)}{row}")
|
|
619
|
+
if left is not None and right is not None and _is_occupied(left) and _is_occupied(right):
|
|
620
|
+
comparisons.append(
|
|
621
|
+
(left.visual_style_id or left.style_id) != (right.visual_style_id or right.style_id)
|
|
622
|
+
)
|
|
623
|
+
return len(comparisons) >= 2 and sum(comparisons) / len(comparisons) >= 0.8
|
|
624
|
+
|
|
625
|
+
|
|
626
|
+
def _is_row_style_boundary(sheet: SheetSnapshot, rect: _Rect, boundary: int) -> bool:
|
|
627
|
+
if boundary - rect.min_row + 1 < 2 or rect.max_row - boundary < 2:
|
|
628
|
+
return False
|
|
629
|
+
if not (
|
|
630
|
+
_rows_have_stable_visual_style(sheet, rect, boundary - 1, boundary)
|
|
631
|
+
and _rows_have_stable_visual_style(sheet, rect, boundary + 1, boundary + 2)
|
|
632
|
+
):
|
|
633
|
+
return False
|
|
634
|
+
comparisons = []
|
|
635
|
+
for column in range(rect.min_column, rect.max_column + 1):
|
|
636
|
+
top = sheet.cells.get(f"{get_column_letter(column)}{boundary}")
|
|
637
|
+
bottom = sheet.cells.get(f"{get_column_letter(column)}{boundary + 1}")
|
|
638
|
+
if top is not None and bottom is not None and _is_occupied(top) and _is_occupied(bottom):
|
|
639
|
+
comparisons.append(
|
|
640
|
+
(top.visual_style_id or top.style_id) != (bottom.visual_style_id or bottom.style_id)
|
|
641
|
+
)
|
|
642
|
+
if len(comparisons) < 2 or sum(comparisons) / len(comparisons) < 0.8:
|
|
643
|
+
return False
|
|
644
|
+
# Only inspect prospective table bodies after finding an actual style
|
|
645
|
+
# transition. Uniform sparse sheets otherwise trigger quadratic scans.
|
|
646
|
+
return _looks_like_complete_table(
|
|
647
|
+
sheet, _Rect(rect.min_column, rect.min_row, rect.max_column, boundary)
|
|
648
|
+
) and _looks_like_complete_table(
|
|
649
|
+
sheet, _Rect(rect.min_column, boundary + 1, rect.max_column, rect.max_row)
|
|
650
|
+
)
|
|
651
|
+
|
|
652
|
+
|
|
653
|
+
def _shared_header_style(sheet: SheetSnapshot, rect: _Rect) -> bool:
|
|
654
|
+
cells = [
|
|
655
|
+
sheet.cells.get(f"{get_column_letter(column)}{rect.min_row}")
|
|
656
|
+
for column in range(rect.min_column, rect.max_column + 1)
|
|
657
|
+
]
|
|
658
|
+
if len(cells) < 2 or any(
|
|
659
|
+
cell is None
|
|
660
|
+
or cell.merge_anchor
|
|
661
|
+
or cell.colspan > 1
|
|
662
|
+
or cell.formula
|
|
663
|
+
or (cell.display_value and not isinstance(cell.raw_value, str))
|
|
664
|
+
for cell in cells
|
|
665
|
+
):
|
|
666
|
+
return False
|
|
667
|
+
styles = Counter(cell.visual_style_id or cell.style_id for cell in cells if cell)
|
|
668
|
+
style, count = styles.most_common(1)[0]
|
|
669
|
+
return (
|
|
670
|
+
bool(style)
|
|
671
|
+
and count / len(cells) >= 0.8
|
|
672
|
+
and sum(bool(cell.display_value.strip()) for cell in cells if cell) >= 2
|
|
673
|
+
and all(
|
|
674
|
+
(cell.visual_style_id or cell.style_id) == style
|
|
675
|
+
for cell in cells
|
|
676
|
+
if cell and not cell.display_value.strip()
|
|
677
|
+
)
|
|
678
|
+
)
|
|
679
|
+
|
|
680
|
+
|
|
681
|
+
def _looks_like_complete_table(sheet: SheetSnapshot, rect: _Rect) -> bool:
|
|
682
|
+
header_cells = [
|
|
683
|
+
cell
|
|
684
|
+
for column in range(rect.min_column, rect.max_column + 1)
|
|
685
|
+
if (cell := sheet.cells.get(f"{get_column_letter(column)}{rect.min_row}")) is not None
|
|
686
|
+
and _is_occupied(cell)
|
|
687
|
+
and cell.merge_anchor is None
|
|
688
|
+
]
|
|
689
|
+
if len(header_cells) < 2 or any(
|
|
690
|
+
cell.formula is not None
|
|
691
|
+
or not isinstance(cell.raw_value, str)
|
|
692
|
+
or not cell.display_value.strip()
|
|
693
|
+
for cell in header_cells
|
|
694
|
+
):
|
|
695
|
+
return False
|
|
696
|
+
# Text-only records are valid table bodies too. Require a populated record
|
|
697
|
+
# under the prospective header, rather than requiring numeric values.
|
|
698
|
+
return any(
|
|
699
|
+
all(
|
|
700
|
+
(cell := sheet.cells.get(f"{get_column_letter(column)}{row}")) is not None
|
|
701
|
+
and _is_occupied(cell)
|
|
702
|
+
and cell.merge_anchor is None
|
|
703
|
+
for column in range(rect.min_column, rect.max_column + 1)
|
|
704
|
+
)
|
|
705
|
+
for row in range(rect.min_row + 1, rect.max_row + 1)
|
|
706
|
+
)
|
|
707
|
+
|
|
708
|
+
|
|
709
|
+
def _rows_have_stable_visual_style(
|
|
710
|
+
sheet: SheetSnapshot,
|
|
711
|
+
rect: _Rect,
|
|
712
|
+
first_row: int,
|
|
713
|
+
second_row: int,
|
|
714
|
+
) -> bool:
|
|
715
|
+
comparisons = []
|
|
716
|
+
for column in range(rect.min_column, rect.max_column + 1):
|
|
717
|
+
first = sheet.cells.get(f"{get_column_letter(column)}{first_row}")
|
|
718
|
+
second = sheet.cells.get(f"{get_column_letter(column)}{second_row}")
|
|
719
|
+
if (
|
|
720
|
+
first is not None
|
|
721
|
+
and second is not None
|
|
722
|
+
and _is_occupied(first)
|
|
723
|
+
and _is_occupied(second)
|
|
724
|
+
):
|
|
725
|
+
comparisons.append(
|
|
726
|
+
(first.visual_style_id or first.style_id)
|
|
727
|
+
== (second.visual_style_id or second.style_id)
|
|
728
|
+
)
|
|
729
|
+
return len(comparisons) >= 2 and sum(comparisons) / len(comparisons) >= 0.8
|
|
730
|
+
|
|
731
|
+
|
|
732
|
+
def _is_density_boundary(
|
|
733
|
+
occupied: dict[str, CellSnapshot],
|
|
734
|
+
positions: dict[str, tuple[int, int]],
|
|
735
|
+
rect: _Rect,
|
|
736
|
+
boundary: int,
|
|
737
|
+
) -> bool:
|
|
738
|
+
left = _Rect(rect.min_column, rect.min_row, boundary, rect.max_row)
|
|
739
|
+
right = _Rect(boundary + 1, rect.min_row, rect.max_column, rect.max_row)
|
|
740
|
+
left_width = left.max_column - left.min_column + 1
|
|
741
|
+
right_width = right.max_column - right.min_column + 1
|
|
742
|
+
if min(left_width, right_width) != 1 or max(left_width, right_width) < 2:
|
|
743
|
+
return False
|
|
744
|
+
left_cells = [
|
|
745
|
+
occupied[coordinate]
|
|
746
|
+
for coordinate, position in positions.items()
|
|
747
|
+
if _contains(left, *position)
|
|
748
|
+
]
|
|
749
|
+
right_cells = [
|
|
750
|
+
occupied[coordinate]
|
|
751
|
+
for coordinate, position in positions.items()
|
|
752
|
+
if _contains(right, *position)
|
|
753
|
+
]
|
|
754
|
+
left_density = len(left_cells) / left.area
|
|
755
|
+
right_density = len(right_cells) / right.area
|
|
756
|
+
dense, sparse = (
|
|
757
|
+
(left_density, right_cells)
|
|
758
|
+
if left_density >= right_density
|
|
759
|
+
else (right_density, left_cells)
|
|
760
|
+
)
|
|
761
|
+
sparse_density = min(left_density, right_density)
|
|
762
|
+
return (
|
|
763
|
+
dense >= 0.75
|
|
764
|
+
and sparse_density <= 0.5
|
|
765
|
+
and any(len(cell.display_value.strip()) >= 20 for cell in sparse)
|
|
766
|
+
)
|
|
767
|
+
|
|
768
|
+
|
|
769
|
+
def _is_row_density_boundary(
|
|
770
|
+
occupied: dict[str, CellSnapshot],
|
|
771
|
+
positions: dict[str, tuple[int, int]],
|
|
772
|
+
rect: _Rect,
|
|
773
|
+
boundary: int,
|
|
774
|
+
) -> bool:
|
|
775
|
+
top = _Rect(rect.min_column, rect.min_row, rect.max_column, boundary)
|
|
776
|
+
bottom = _Rect(rect.min_column, boundary + 1, rect.max_column, rect.max_row)
|
|
777
|
+
top_height = top.max_row - top.min_row + 1
|
|
778
|
+
bottom_height = bottom.max_row - bottom.min_row + 1
|
|
779
|
+
if min(top_height, bottom_height) != 1 or max(top_height, bottom_height) < 2:
|
|
780
|
+
return False
|
|
781
|
+
top_cells = [
|
|
782
|
+
occupied[coordinate]
|
|
783
|
+
for coordinate, position in positions.items()
|
|
784
|
+
if _contains(top, *position)
|
|
785
|
+
]
|
|
786
|
+
bottom_cells = [
|
|
787
|
+
occupied[coordinate]
|
|
788
|
+
for coordinate, position in positions.items()
|
|
789
|
+
if _contains(bottom, *position)
|
|
790
|
+
]
|
|
791
|
+
top_density = len(top_cells) / top.area
|
|
792
|
+
bottom_density = len(bottom_cells) / bottom.area
|
|
793
|
+
dense, sparse = (
|
|
794
|
+
(top_density, bottom_cells)
|
|
795
|
+
if top_density >= bottom_density
|
|
796
|
+
else (bottom_density, top_cells)
|
|
797
|
+
)
|
|
798
|
+
sparse_density = min(top_density, bottom_density)
|
|
799
|
+
return (
|
|
800
|
+
dense >= 0.75
|
|
801
|
+
and sparse_density <= 0.5
|
|
802
|
+
and any(len(cell.display_value.strip()) >= 20 for cell in sparse)
|
|
803
|
+
)
|
|
804
|
+
|
|
805
|
+
|
|
806
|
+
def _formula_boundaries(
|
|
807
|
+
occupied: dict[str, CellSnapshot],
|
|
808
|
+
positions: dict[str, tuple[int, int]],
|
|
809
|
+
rect: _Rect,
|
|
810
|
+
) -> tuple[set[int], set[int]]:
|
|
811
|
+
columns: set[int] = set()
|
|
812
|
+
rows: set[int] = set()
|
|
813
|
+
for coordinate, cell in occupied.items():
|
|
814
|
+
if cell.formula is None:
|
|
815
|
+
continue
|
|
816
|
+
row, column = positions[coordinate]
|
|
817
|
+
if not _contains(rect, row, column):
|
|
818
|
+
continue
|
|
819
|
+
for ref_column, ref_row in _local_formula_refs(cell.formula):
|
|
820
|
+
if _contains(rect, ref_row, ref_column):
|
|
821
|
+
columns.update(range(min(column, ref_column), max(column, ref_column)))
|
|
822
|
+
rows.update(range(min(row, ref_row), max(row, ref_row)))
|
|
823
|
+
return columns, rows
|
|
824
|
+
|
|
825
|
+
|
|
826
|
+
def _local_formula_refs(formula: str) -> list[tuple[int, int]]:
|
|
827
|
+
try:
|
|
828
|
+
range_tokens = (
|
|
829
|
+
token.value
|
|
830
|
+
for token in Tokenizer(formula).items
|
|
831
|
+
if token.type == "OPERAND" and token.subtype == "RANGE" and "!" not in token.value
|
|
832
|
+
)
|
|
833
|
+
except Exception:
|
|
834
|
+
return []
|
|
835
|
+
return [
|
|
836
|
+
(column_index_from_string(column_name), int(row_number))
|
|
837
|
+
for value in range_tokens
|
|
838
|
+
for column_name, row_number in _FORMULA_CELL_REF.findall(value.upper())
|
|
839
|
+
]
|
|
840
|
+
|
|
841
|
+
|
|
842
|
+
def _merged_range_crosses_column(sheet: SheetSnapshot, rect: _Rect, boundary: int) -> bool:
|
|
843
|
+
return any(
|
|
844
|
+
merged.min_column <= boundary < merged.max_column
|
|
845
|
+
and not (merged.max_row < rect.min_row or merged.min_row > rect.max_row)
|
|
846
|
+
for merged in _merged_rects(sheet)
|
|
847
|
+
)
|
|
848
|
+
|
|
849
|
+
|
|
850
|
+
def _merged_range_crosses_row(sheet: SheetSnapshot, rect: _Rect, boundary: int) -> bool:
|
|
851
|
+
return any(
|
|
852
|
+
merged.min_row <= boundary < merged.max_row
|
|
853
|
+
and not (merged.max_column < rect.min_column or merged.min_column > rect.max_column)
|
|
854
|
+
for merged in _merged_rects(sheet)
|
|
855
|
+
)
|
|
856
|
+
|
|
857
|
+
|
|
858
|
+
def _merged_rects(sheet: SheetSnapshot) -> list[_Rect]:
|
|
859
|
+
rectangles = []
|
|
860
|
+
for value in sheet.merged_ranges:
|
|
861
|
+
try:
|
|
862
|
+
rectangles.append(_rect_from_range(value.replace("$", "")))
|
|
863
|
+
except ValueError:
|
|
864
|
+
continue
|
|
865
|
+
return rectangles
|
|
866
|
+
|
|
867
|
+
|
|
868
|
+
def _anchor_reason(kind: str) -> str:
|
|
869
|
+
return "native_table_anchor" if kind == "excel_table" else "defined_name_anchor"
|
|
870
|
+
|
|
871
|
+
|
|
872
|
+
def _rect_from_range(value: str) -> _Rect:
|
|
873
|
+
boundaries = range_boundaries(value)
|
|
874
|
+
if any(boundary is None for boundary in boundaries):
|
|
875
|
+
raise ValueError("region ranges must have finite row and column bounds")
|
|
876
|
+
min_column, min_row, max_column, max_row = boundaries
|
|
877
|
+
return _Rect(min_column, min_row, max_column, max_row)
|
|
878
|
+
|
|
879
|
+
|
|
880
|
+
def _contains(rect: _Rect, row: int, column: int) -> bool:
|
|
881
|
+
return rect.min_row <= row <= rect.max_row and rect.min_column <= column <= rect.max_column
|
|
882
|
+
|
|
883
|
+
|
|
884
|
+
def _rect_inside(inner: _Rect, outer: _Rect) -> bool:
|
|
885
|
+
return (
|
|
886
|
+
outer.min_column <= inner.min_column <= inner.max_column <= outer.max_column
|
|
887
|
+
and outer.min_row <= inner.min_row <= inner.max_row <= outer.max_row
|
|
888
|
+
)
|
|
889
|
+
|
|
890
|
+
|
|
891
|
+
def _rectangles_overlap(left: _Rect, right: _Rect) -> bool:
|
|
892
|
+
return not (
|
|
893
|
+
left.max_column < right.min_column
|
|
894
|
+
or right.max_column < left.min_column
|
|
895
|
+
or left.max_row < right.min_row
|
|
896
|
+
or right.max_row < left.min_row
|
|
897
|
+
)
|
|
898
|
+
|
|
899
|
+
|
|
900
|
+
def _pairwise_non_overlapping(rectangles: list[_Rect]) -> bool:
|
|
901
|
+
return all(
|
|
902
|
+
not _rectangles_overlap(left, right)
|
|
903
|
+
for index, left in enumerate(rectangles)
|
|
904
|
+
for right in rectangles[index + 1 :]
|
|
905
|
+
)
|
|
906
|
+
|
|
907
|
+
|
|
908
|
+
def _is_occupied(cell: CellSnapshot) -> bool:
|
|
909
|
+
return any(
|
|
910
|
+
(
|
|
911
|
+
cell.raw_value is not None,
|
|
912
|
+
cell.formula is not None,
|
|
913
|
+
cell.comment is not None,
|
|
914
|
+
cell.hyperlink is not None,
|
|
915
|
+
cell.merge_anchor is not None,
|
|
916
|
+
)
|
|
917
|
+
)
|
|
918
|
+
|
|
919
|
+
|
|
920
|
+
def _consecutive_groups(values: Iterable[int]) -> list[tuple[int, int]]:
|
|
921
|
+
ordered = sorted(set(values))
|
|
922
|
+
if not ordered:
|
|
923
|
+
return []
|
|
924
|
+
groups: list[tuple[int, int]] = []
|
|
925
|
+
start = previous = ordered[0]
|
|
926
|
+
for value in ordered[1:]:
|
|
927
|
+
if value != previous + 1:
|
|
928
|
+
groups.append((start, previous))
|
|
929
|
+
start = value
|
|
930
|
+
previous = value
|
|
931
|
+
groups.append((start, previous))
|
|
932
|
+
return groups
|