langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,932 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ from collections import Counter
5
+ from collections.abc import Iterable
6
+ from dataclasses import dataclass
7
+ from typing import Any
8
+
9
+ from openpyxl.formula import Tokenizer
10
+ from openpyxl.utils import column_index_from_string, get_column_letter, range_boundaries
11
+ from openpyxl.utils.cell import coordinate_to_tuple
12
+
13
+ from langparse.workbooks.labels import is_section_label, is_total_label
14
+ from langparse.workbooks.types import (
15
+ CandidateRegion,
16
+ CellSnapshot,
17
+ RegionAnchor,
18
+ SheetSnapshot,
19
+ SourceRef,
20
+ )
21
+
22
+ _FORMULA_CELL_REF = re.compile(r"(?<![A-Z0-9_!])\$?([A-Z]{1,3})\$?([1-9][0-9]*)(?![A-Z0-9_(])")
23
+ _REASON_ORDER = {
24
+ "native_table_anchor": 0,
25
+ "defined_name_anchor": 1,
26
+ "print_area_anchor": 2,
27
+ "merged_title_anchor": 3,
28
+ "formula_continuity": 4,
29
+ "style_boundary": 5,
30
+ "density_boundary": 6,
31
+ "blank_band": 7,
32
+ "occupied_extent": 8,
33
+ }
34
+ _EXACT_ANCHOR_REASONS = {
35
+ "native_table_anchor",
36
+ "defined_name_anchor",
37
+ "print_area_anchor",
38
+ }
39
+
40
+
41
+ @dataclass(frozen=True)
42
+ class _Rect:
43
+ min_column: int
44
+ min_row: int
45
+ max_column: int
46
+ max_row: int
47
+
48
+ @property
49
+ def area(self) -> int:
50
+ return (self.max_column - self.min_column + 1) * (self.max_row - self.min_row + 1)
51
+
52
+ @property
53
+ def range(self) -> str:
54
+ return (
55
+ f"{get_column_letter(self.min_column)}{self.min_row}:"
56
+ f"{get_column_letter(self.max_column)}{self.max_row}"
57
+ )
58
+
59
+
60
+ @dataclass(frozen=True)
61
+ class _UsableAnchor:
62
+ anchor: RegionAnchor
63
+ rect: _Rect
64
+
65
+
66
+ @dataclass(frozen=True)
67
+ class _Cut:
68
+ orientation: str
69
+ boundary: int
70
+ reason: str
71
+
72
+
73
+ def detect_candidate_regions(sheet: SheetSnapshot) -> list[CandidateRegion]:
74
+ """Partition assignable cells using stable structural evidence.
75
+
76
+ The public interface intentionally remains a single deterministic function.
77
+ Native anchors, print areas, visual discontinuities, merged ranges and formula
78
+ references are implementation details hidden behind that seam.
79
+ """
80
+
81
+ occupied = {coordinate: cell for coordinate, cell in sheet.cells.items() if _is_occupied(cell)}
82
+ if not occupied:
83
+ return []
84
+
85
+ positions = {coordinate: coordinate_to_tuple(coordinate) for coordinate in occupied}
86
+ coarse_regions = _initial_coarse_regions(sheet, positions)
87
+ blank_partitioned = len(coarse_regions) > 1
88
+ regions: list[CandidateRegion] = []
89
+ for coarse in coarse_regions:
90
+ regions.extend(
91
+ _partition_coarse_region(
92
+ sheet,
93
+ occupied,
94
+ positions,
95
+ coarse,
96
+ blank_partitioned=blank_partitioned,
97
+ )
98
+ )
99
+
100
+ regions.sort(
101
+ key=lambda region: (
102
+ coordinate_to_tuple(region.source_ref.range.split(":", 1)[0]),
103
+ region.source_ref.range,
104
+ )
105
+ )
106
+ assigned = [coordinate for region in regions for coordinate in region.cell_refs]
107
+ if Counter(assigned) != Counter(occupied.keys()):
108
+ raise RuntimeError("Candidate region partition violated cell ownership")
109
+ return regions
110
+
111
+
112
+ def _initial_coarse_regions(
113
+ sheet: SheetSnapshot,
114
+ positions: dict[str, tuple[int, int]],
115
+ ) -> list[_Rect]:
116
+ coarse_regions = []
117
+ for min_row, max_row in _consecutive_groups(row for row, _ in positions.values()):
118
+ columns = {column for row, column in positions.values() if min_row <= row <= max_row}
119
+ band = _Rect(min(columns), min_row, max(columns), max_row)
120
+ if _shared_header_style(sheet, band):
121
+ # A styled header extending through blank input columns supplies
122
+ # positive continuity evidence; blank cells remain non-assignable.
123
+ coarse_regions.append(band)
124
+ else:
125
+ coarse_regions.extend(
126
+ _Rect(min_column, min_row, max_column, max_row)
127
+ for min_column, max_column in _consecutive_groups(columns)
128
+ )
129
+
130
+ coarse_regions = _join_section_bands(sheet, coarse_regions)
131
+
132
+ for anchor in sheet.region_anchors:
133
+ if anchor.kind != "excel_table" or anchor.source_ref.sheet_name != sheet.name:
134
+ continue
135
+ try:
136
+ anchor_rect = _rect_from_range(anchor.source_ref.range)
137
+ except ValueError:
138
+ continue
139
+ if not any(_contains(anchor_rect, *position) for position in positions.values()):
140
+ continue
141
+ overlapping = [rect for rect in coarse_regions if _rectangles_overlap(rect, anchor_rect)]
142
+ if not overlapping:
143
+ continue
144
+ coarse_regions = [
145
+ rect for rect in coarse_regions if not _rectangles_overlap(rect, anchor_rect)
146
+ ]
147
+ coarse_regions.append(_bounding_rect([anchor_rect, *overlapping]))
148
+ coarse_regions = _merge_overlapping_rects(coarse_regions)
149
+
150
+ return sorted(
151
+ coarse_regions,
152
+ key=lambda rect: (rect.min_row, rect.min_column, rect.max_row, rect.max_column),
153
+ )
154
+
155
+
156
+ def _join_section_bands(sheet: SheetSnapshot, rectangles: list[_Rect]) -> list[_Rect]:
157
+ result: list[_Rect] = []
158
+ for rect in rectangles:
159
+ previous = result[-1] if result else None
160
+ first = sheet.cells.get(f"{get_column_letter(rect.min_column)}{rect.min_row}")
161
+ continuation = first is not None and (
162
+ (first.colspan > 1 and is_section_label(first.display_value))
163
+ or is_total_label(first.display_value)
164
+ )
165
+ if (
166
+ previous is not None
167
+ and continuation
168
+ and previous.min_column == rect.min_column
169
+ and rect.max_column <= previous.max_column
170
+ and 0 < rect.min_row - previous.max_row <= 2
171
+ and _shared_header_style(sheet, previous)
172
+ ):
173
+ # Include the whole continuation row band, including disconnected
174
+ # amount cells following a merged subtotal label.
175
+ band_end = rect.max_row
176
+ result[-1] = _Rect(previous.min_column, previous.min_row, previous.max_column, band_end)
177
+ elif previous is not None and (
178
+ previous.min_row <= rect.min_row <= rect.max_row <= previous.max_row
179
+ and previous.min_column <= rect.min_column <= rect.max_column <= previous.max_column
180
+ ):
181
+ continue
182
+ else:
183
+ result.append(rect)
184
+ return result
185
+
186
+
187
+ def _bounding_rect(rectangles: list[_Rect]) -> _Rect:
188
+ return _Rect(
189
+ min(rect.min_column for rect in rectangles),
190
+ min(rect.min_row for rect in rectangles),
191
+ max(rect.max_column for rect in rectangles),
192
+ max(rect.max_row for rect in rectangles),
193
+ )
194
+
195
+
196
+ def _merge_overlapping_rects(rectangles: list[_Rect]) -> list[_Rect]:
197
+ pending = list(rectangles)
198
+ merged: list[_Rect] = []
199
+ while pending:
200
+ current = pending.pop()
201
+ overlaps = [rect for rect in pending if _rectangles_overlap(current, rect)]
202
+ if overlaps:
203
+ pending = [rect for rect in pending if rect not in overlaps]
204
+ pending.append(_bounding_rect([current, *overlaps]))
205
+ else:
206
+ merged.append(current)
207
+ return merged
208
+
209
+
210
+ def _partition_coarse_region(
211
+ sheet: SheetSnapshot,
212
+ occupied: dict[str, CellSnapshot],
213
+ positions: dict[str, tuple[int, int]],
214
+ coarse: _Rect,
215
+ *,
216
+ blank_partitioned: bool,
217
+ ) -> list[CandidateRegion]:
218
+ selected_anchors, conflicts = _select_anchors(sheet, positions, coarse)
219
+ print_rects = _print_area_rects(sheet, coarse)
220
+ protected_rects = [selected.rect for selected in selected_anchors]
221
+ use_print_rects = len(print_rects) >= 2 and _pairwise_non_overlapping(print_rects)
222
+ if use_print_rects:
223
+ protected_rects.extend(print_rects)
224
+ return _partition_rect(
225
+ sheet,
226
+ occupied,
227
+ positions,
228
+ coarse,
229
+ selected_anchors,
230
+ conflicts,
231
+ print_rects,
232
+ use_print_rects=use_print_rects,
233
+ protected_rects=protected_rects,
234
+ partition_reasons=frozenset(),
235
+ blank_partitioned=blank_partitioned,
236
+ )
237
+
238
+
239
+ def _partition_rect(
240
+ sheet: SheetSnapshot,
241
+ occupied: dict[str, CellSnapshot],
242
+ positions: dict[str, tuple[int, int]],
243
+ rect: _Rect,
244
+ selected_anchors: list[_UsableAnchor],
245
+ conflicts: list[tuple[_UsableAnchor, _UsableAnchor]],
246
+ print_rects: list[_Rect],
247
+ *,
248
+ use_print_rects: bool,
249
+ protected_rects: list[_Rect],
250
+ partition_reasons: frozenset[str],
251
+ blank_partitioned: bool,
252
+ ) -> list[CandidateRegion]:
253
+ cell_refs = sorted(
254
+ (coordinate for coordinate, position in positions.items() if _contains(rect, *position)),
255
+ key=coordinate_to_tuple,
256
+ )
257
+ if not cell_refs:
258
+ return []
259
+
260
+ cuts = _candidate_cuts(
261
+ sheet,
262
+ occupied,
263
+ positions,
264
+ rect,
265
+ selected_anchors,
266
+ print_rects,
267
+ use_print_rects=use_print_rects,
268
+ protected_rects=protected_rects,
269
+ )
270
+ if cuts:
271
+ cut = cuts[0]
272
+ first, second = _split_rect(rect, cut)
273
+ inherited_reasons = partition_reasons
274
+ if cut.reason not in _EXACT_ANCHOR_REASONS:
275
+ inherited_reasons = frozenset((*partition_reasons, cut.reason))
276
+ return [
277
+ *_partition_rect(
278
+ sheet,
279
+ occupied,
280
+ positions,
281
+ first,
282
+ selected_anchors,
283
+ conflicts,
284
+ print_rects,
285
+ use_print_rects=use_print_rects,
286
+ protected_rects=protected_rects,
287
+ partition_reasons=inherited_reasons,
288
+ blank_partitioned=blank_partitioned,
289
+ ),
290
+ *_partition_rect(
291
+ sheet,
292
+ occupied,
293
+ positions,
294
+ second,
295
+ selected_anchors,
296
+ conflicts,
297
+ print_rects,
298
+ use_print_rects=use_print_rects,
299
+ protected_rects=protected_rects,
300
+ partition_reasons=inherited_reasons,
301
+ blank_partitioned=blank_partitioned,
302
+ ),
303
+ ]
304
+
305
+ source_rect = _anchored_source_rect(rect, cell_refs, positions, selected_anchors)
306
+ reasons = _region_reasons(
307
+ sheet,
308
+ occupied,
309
+ positions,
310
+ source_rect,
311
+ selected_anchors,
312
+ print_rects,
313
+ partition_reasons=partition_reasons,
314
+ blank_partitioned=blank_partitioned,
315
+ )
316
+ diagnostics = [
317
+ {
318
+ "reason_code": "overlapping_native_anchors",
319
+ "kept_kind": kept.anchor.kind,
320
+ "kept_range": kept.rect.range,
321
+ "kept_name": kept.anchor.name,
322
+ "kept_scope": kept.anchor.scope,
323
+ "rejected_kind": rejected.anchor.kind,
324
+ "rejected_range": rejected.rect.range,
325
+ "rejected_name": rejected.anchor.name,
326
+ "rejected_scope": rejected.anchor.scope,
327
+ }
328
+ for kept, rejected in conflicts
329
+ if _rectangles_overlap(source_rect, kept.rect)
330
+ or _rectangles_overlap(source_rect, rejected.rect)
331
+ ]
332
+ return [
333
+ CandidateRegion(
334
+ source_ref=SourceRef(sheet_name=sheet.name, range=source_rect.range),
335
+ cell_refs=cell_refs,
336
+ confidence=_region_confidence(reasons, diagnostics),
337
+ features={
338
+ "row_count": source_rect.max_row - source_rect.min_row + 1,
339
+ "column_count": source_rect.max_column - source_rect.min_column + 1,
340
+ "occupied_count": len(cell_refs),
341
+ "density": len(cell_refs) / source_rect.area,
342
+ },
343
+ diagnostics=diagnostics,
344
+ reason_codes=reasons,
345
+ )
346
+ ]
347
+
348
+
349
+ def _candidate_cuts(
350
+ sheet: SheetSnapshot,
351
+ occupied: dict[str, CellSnapshot],
352
+ positions: dict[str, tuple[int, int]],
353
+ rect: _Rect,
354
+ selected_anchors: list[_UsableAnchor],
355
+ print_rects: list[_Rect],
356
+ *,
357
+ use_print_rects: bool,
358
+ protected_rects: list[_Rect],
359
+ ) -> list[_Cut]:
360
+ cuts: dict[tuple[str, int], _Cut] = {}
361
+ # Formula-free exports are common. Never scan every cell for every possible
362
+ # row cut; tokenize each formula once and index the boundaries it protects.
363
+ formula_columns, formula_rows = _formula_boundaries(occupied, positions, rect)
364
+ evidence = [
365
+ (selected.rect, _anchor_reason(selected.anchor.kind))
366
+ for selected in selected_anchors
367
+ if _rectangles_overlap(selected.rect, rect)
368
+ ]
369
+ if use_print_rects:
370
+ evidence.extend(
371
+ (print_rect, "print_area_anchor")
372
+ for print_rect in print_rects
373
+ if _rectangles_overlap(print_rect, rect)
374
+ )
375
+ for evidence_rect, reason in evidence:
376
+ for cut in _rect_edge_cuts(evidence_rect, rect, reason):
377
+ if _cut_is_safe(sheet, rect, cut, protected_rects):
378
+ _offer_cut(cuts, cut)
379
+
380
+ for boundary in range(rect.min_column, rect.max_column):
381
+ key = ("vertical", boundary)
382
+ if key in cuts:
383
+ continue
384
+ cut = _Cut("vertical", boundary, "style_boundary")
385
+ if not _cut_is_safe(sheet, rect, cut, protected_rects):
386
+ continue
387
+ if boundary in formula_columns:
388
+ continue
389
+ if _is_style_boundary(sheet, rect, boundary):
390
+ cuts[key] = cut
391
+ elif _is_density_boundary(occupied, positions, rect, boundary):
392
+ cuts[key] = _Cut("vertical", boundary, "density_boundary")
393
+
394
+ for boundary in range(rect.min_row, rect.max_row):
395
+ key = ("horizontal", boundary)
396
+ if key in cuts:
397
+ continue
398
+ cut = _Cut("horizontal", boundary, "style_boundary")
399
+ if not _cut_is_safe(sheet, rect, cut, protected_rects):
400
+ continue
401
+ if boundary in formula_rows:
402
+ continue
403
+ if _is_row_style_boundary(sheet, rect, boundary):
404
+ cuts[key] = cut
405
+ elif _is_row_density_boundary(occupied, positions, rect, boundary):
406
+ cuts[key] = _Cut("horizontal", boundary, "density_boundary")
407
+
408
+ return sorted(
409
+ cuts.values(),
410
+ key=lambda cut: (
411
+ _REASON_ORDER.get(cut.reason, 99),
412
+ 0 if cut.orientation == "vertical" else 1,
413
+ cut.boundary,
414
+ ),
415
+ )
416
+
417
+
418
+ def _rect_edge_cuts(evidence: _Rect, rect: _Rect, reason: str) -> list[_Cut]:
419
+ cuts = []
420
+ if rect.min_column < evidence.min_column <= rect.max_column:
421
+ cuts.append(_Cut("vertical", evidence.min_column - 1, reason))
422
+ if rect.min_column <= evidence.max_column < rect.max_column:
423
+ cuts.append(_Cut("vertical", evidence.max_column, reason))
424
+ if rect.min_row < evidence.min_row <= rect.max_row:
425
+ cuts.append(_Cut("horizontal", evidence.min_row - 1, reason))
426
+ if rect.min_row <= evidence.max_row < rect.max_row:
427
+ cuts.append(_Cut("horizontal", evidence.max_row, reason))
428
+ return cuts
429
+
430
+
431
+ def _offer_cut(cuts: dict[tuple[str, int], _Cut], candidate: _Cut) -> None:
432
+ key = (candidate.orientation, candidate.boundary)
433
+ current = cuts.get(key)
434
+ if current is None or _REASON_ORDER.get(candidate.reason, 99) < _REASON_ORDER.get(
435
+ current.reason,
436
+ 99,
437
+ ):
438
+ cuts[key] = candidate
439
+
440
+
441
+ def _cut_is_safe(
442
+ sheet: SheetSnapshot,
443
+ rect: _Rect,
444
+ cut: _Cut,
445
+ protected_rects: list[_Rect],
446
+ ) -> bool:
447
+ if cut.orientation == "vertical":
448
+ if _merged_range_crosses_column(sheet, rect, cut.boundary):
449
+ return False
450
+ return not any(
451
+ _rectangles_overlap(protected, rect)
452
+ and protected.min_column <= cut.boundary < protected.max_column
453
+ for protected in protected_rects
454
+ )
455
+ if _merged_range_crosses_row(sheet, rect, cut.boundary):
456
+ return False
457
+ return not any(
458
+ _rectangles_overlap(protected, rect)
459
+ and protected.min_row <= cut.boundary < protected.max_row
460
+ for protected in protected_rects
461
+ )
462
+
463
+
464
+ def _split_rect(rect: _Rect, cut: _Cut) -> tuple[_Rect, _Rect]:
465
+ if cut.orientation == "vertical":
466
+ return (
467
+ _Rect(rect.min_column, rect.min_row, cut.boundary, rect.max_row),
468
+ _Rect(cut.boundary + 1, rect.min_row, rect.max_column, rect.max_row),
469
+ )
470
+ return (
471
+ _Rect(rect.min_column, rect.min_row, rect.max_column, cut.boundary),
472
+ _Rect(rect.min_column, cut.boundary + 1, rect.max_column, rect.max_row),
473
+ )
474
+
475
+
476
+ def _select_anchors(
477
+ sheet: SheetSnapshot,
478
+ positions: dict[str, tuple[int, int]],
479
+ coarse: _Rect,
480
+ ) -> tuple[list[_UsableAnchor], list[tuple[_UsableAnchor, _UsableAnchor]]]:
481
+ usable = []
482
+ for anchor in sheet.region_anchors:
483
+ if anchor.source_ref.sheet_name != sheet.name:
484
+ continue
485
+ try:
486
+ rect = _rect_from_range(anchor.source_ref.range)
487
+ except ValueError:
488
+ continue
489
+ if not _rectangles_overlap(rect, coarse) or not _rect_inside(rect, coarse):
490
+ continue
491
+ count = sum(_contains(rect, row, column) for row, column in positions.values())
492
+ if not count:
493
+ continue
494
+ if anchor.kind == "defined_name" and (
495
+ rect.max_column == rect.min_column
496
+ or rect.max_row == rect.min_row
497
+ or count / rect.area < 0.5
498
+ ):
499
+ continue
500
+ if anchor.kind not in {"excel_table", "defined_name"}:
501
+ continue
502
+ usable.append(_UsableAnchor(anchor, rect))
503
+
504
+ usable.sort(
505
+ key=lambda item: (
506
+ {"excel_table": 0, "defined_name": 1}.get(item.anchor.kind, 99),
507
+ -item.rect.area,
508
+ item.rect.min_row,
509
+ item.rect.min_column,
510
+ item.anchor.name or "",
511
+ )
512
+ )
513
+ selected: list[_UsableAnchor] = []
514
+ conflicts: list[tuple[_UsableAnchor, _UsableAnchor]] = []
515
+ for candidate in usable:
516
+ conflicting = next(
517
+ (item for item in selected if _rectangles_overlap(item.rect, candidate.rect)),
518
+ None,
519
+ )
520
+ if conflicting is not None:
521
+ if conflicting.rect == candidate.rect:
522
+ continue
523
+ conflicts.append((conflicting, candidate))
524
+ continue
525
+ selected.append(candidate)
526
+ return selected, conflicts
527
+
528
+
529
+ def _print_area_rects(sheet: SheetSnapshot, coarse: _Rect) -> list[_Rect]:
530
+ rectangles = []
531
+ for value in sheet.print_area:
532
+ local_range = value.rsplit("!", 1)[-1].replace("$", "")
533
+ try:
534
+ rect = _rect_from_range(local_range)
535
+ except ValueError:
536
+ continue
537
+ if _rect_inside(rect, coarse):
538
+ rectangles.append(rect)
539
+ return sorted(
540
+ set(rectangles),
541
+ key=lambda item: (item.min_row, item.min_column, item.max_row, item.max_column),
542
+ )
543
+
544
+
545
+ def _anchored_source_rect(
546
+ segment_rect: _Rect,
547
+ cell_refs: list[str],
548
+ positions: dict[str, tuple[int, int]],
549
+ anchors: list[_UsableAnchor],
550
+ ) -> _Rect:
551
+ for selected in anchors:
552
+ anchor_cells = {
553
+ coordinate
554
+ for coordinate, position in positions.items()
555
+ if _contains(selected.rect, *position)
556
+ }
557
+ if anchor_cells == set(cell_refs) and _rect_inside(selected.rect, segment_rect):
558
+ return selected.rect
559
+ return segment_rect
560
+
561
+
562
+ def _region_reasons(
563
+ sheet: SheetSnapshot,
564
+ occupied: dict[str, CellSnapshot],
565
+ positions: dict[str, tuple[int, int]],
566
+ rect: _Rect,
567
+ anchors: list[_UsableAnchor],
568
+ print_rects: list[_Rect],
569
+ *,
570
+ partition_reasons: frozenset[str],
571
+ blank_partitioned: bool,
572
+ ) -> list[str]:
573
+ reasons = set(partition_reasons)
574
+ for selected in anchors:
575
+ if selected.rect == rect:
576
+ reasons.add(_anchor_reason(selected.anchor.kind))
577
+ if rect in print_rects:
578
+ reasons.add("print_area_anchor")
579
+ if any(
580
+ cell.formula is not None and _local_formula_refs(cell.formula)
581
+ for coordinate, cell in occupied.items()
582
+ if _contains(rect, *positions[coordinate])
583
+ ):
584
+ reasons.add("formula_continuity")
585
+ if any(_rectangles_overlap(rect, merged) for merged in _merged_rects(sheet)):
586
+ reasons.add("merged_title_anchor")
587
+ if blank_partitioned:
588
+ reasons.add("blank_band")
589
+ if not reasons or reasons == {"formula_continuity"}:
590
+ reasons.add("occupied_extent")
591
+ return sorted(reasons, key=lambda item: (_REASON_ORDER.get(item, 99), item))
592
+
593
+
594
+ def _region_confidence(reasons: list[str], diagnostics: list[dict[str, Any]]) -> float:
595
+ if "native_table_anchor" in reasons:
596
+ confidence = 0.98
597
+ elif "print_area_anchor" in reasons or "defined_name_anchor" in reasons:
598
+ confidence = 0.95
599
+ elif "style_boundary" in reasons:
600
+ confidence = 0.9
601
+ elif "density_boundary" in reasons:
602
+ confidence = 0.82
603
+ else:
604
+ confidence = 1.0
605
+ if diagnostics:
606
+ confidence = min(confidence, 0.75)
607
+ return confidence
608
+
609
+
610
+ def _is_style_boundary(sheet: SheetSnapshot, rect: _Rect, boundary: int) -> bool:
611
+ if boundary - rect.min_column + 1 < 2 or rect.max_column - boundary < 2:
612
+ return False
613
+ if _shared_header_style(sheet, rect):
614
+ return False
615
+ comparisons = []
616
+ for row in range(rect.min_row, rect.max_row + 1):
617
+ left = sheet.cells.get(f"{get_column_letter(boundary)}{row}")
618
+ right = sheet.cells.get(f"{get_column_letter(boundary + 1)}{row}")
619
+ if left is not None and right is not None and _is_occupied(left) and _is_occupied(right):
620
+ comparisons.append(
621
+ (left.visual_style_id or left.style_id) != (right.visual_style_id or right.style_id)
622
+ )
623
+ return len(comparisons) >= 2 and sum(comparisons) / len(comparisons) >= 0.8
624
+
625
+
626
+ def _is_row_style_boundary(sheet: SheetSnapshot, rect: _Rect, boundary: int) -> bool:
627
+ if boundary - rect.min_row + 1 < 2 or rect.max_row - boundary < 2:
628
+ return False
629
+ if not (
630
+ _rows_have_stable_visual_style(sheet, rect, boundary - 1, boundary)
631
+ and _rows_have_stable_visual_style(sheet, rect, boundary + 1, boundary + 2)
632
+ ):
633
+ return False
634
+ comparisons = []
635
+ for column in range(rect.min_column, rect.max_column + 1):
636
+ top = sheet.cells.get(f"{get_column_letter(column)}{boundary}")
637
+ bottom = sheet.cells.get(f"{get_column_letter(column)}{boundary + 1}")
638
+ if top is not None and bottom is not None and _is_occupied(top) and _is_occupied(bottom):
639
+ comparisons.append(
640
+ (top.visual_style_id or top.style_id) != (bottom.visual_style_id or bottom.style_id)
641
+ )
642
+ if len(comparisons) < 2 or sum(comparisons) / len(comparisons) < 0.8:
643
+ return False
644
+ # Only inspect prospective table bodies after finding an actual style
645
+ # transition. Uniform sparse sheets otherwise trigger quadratic scans.
646
+ return _looks_like_complete_table(
647
+ sheet, _Rect(rect.min_column, rect.min_row, rect.max_column, boundary)
648
+ ) and _looks_like_complete_table(
649
+ sheet, _Rect(rect.min_column, boundary + 1, rect.max_column, rect.max_row)
650
+ )
651
+
652
+
653
+ def _shared_header_style(sheet: SheetSnapshot, rect: _Rect) -> bool:
654
+ cells = [
655
+ sheet.cells.get(f"{get_column_letter(column)}{rect.min_row}")
656
+ for column in range(rect.min_column, rect.max_column + 1)
657
+ ]
658
+ if len(cells) < 2 or any(
659
+ cell is None
660
+ or cell.merge_anchor
661
+ or cell.colspan > 1
662
+ or cell.formula
663
+ or (cell.display_value and not isinstance(cell.raw_value, str))
664
+ for cell in cells
665
+ ):
666
+ return False
667
+ styles = Counter(cell.visual_style_id or cell.style_id for cell in cells if cell)
668
+ style, count = styles.most_common(1)[0]
669
+ return (
670
+ bool(style)
671
+ and count / len(cells) >= 0.8
672
+ and sum(bool(cell.display_value.strip()) for cell in cells if cell) >= 2
673
+ and all(
674
+ (cell.visual_style_id or cell.style_id) == style
675
+ for cell in cells
676
+ if cell and not cell.display_value.strip()
677
+ )
678
+ )
679
+
680
+
681
+ def _looks_like_complete_table(sheet: SheetSnapshot, rect: _Rect) -> bool:
682
+ header_cells = [
683
+ cell
684
+ for column in range(rect.min_column, rect.max_column + 1)
685
+ if (cell := sheet.cells.get(f"{get_column_letter(column)}{rect.min_row}")) is not None
686
+ and _is_occupied(cell)
687
+ and cell.merge_anchor is None
688
+ ]
689
+ if len(header_cells) < 2 or any(
690
+ cell.formula is not None
691
+ or not isinstance(cell.raw_value, str)
692
+ or not cell.display_value.strip()
693
+ for cell in header_cells
694
+ ):
695
+ return False
696
+ # Text-only records are valid table bodies too. Require a populated record
697
+ # under the prospective header, rather than requiring numeric values.
698
+ return any(
699
+ all(
700
+ (cell := sheet.cells.get(f"{get_column_letter(column)}{row}")) is not None
701
+ and _is_occupied(cell)
702
+ and cell.merge_anchor is None
703
+ for column in range(rect.min_column, rect.max_column + 1)
704
+ )
705
+ for row in range(rect.min_row + 1, rect.max_row + 1)
706
+ )
707
+
708
+
709
+ def _rows_have_stable_visual_style(
710
+ sheet: SheetSnapshot,
711
+ rect: _Rect,
712
+ first_row: int,
713
+ second_row: int,
714
+ ) -> bool:
715
+ comparisons = []
716
+ for column in range(rect.min_column, rect.max_column + 1):
717
+ first = sheet.cells.get(f"{get_column_letter(column)}{first_row}")
718
+ second = sheet.cells.get(f"{get_column_letter(column)}{second_row}")
719
+ if (
720
+ first is not None
721
+ and second is not None
722
+ and _is_occupied(first)
723
+ and _is_occupied(second)
724
+ ):
725
+ comparisons.append(
726
+ (first.visual_style_id or first.style_id)
727
+ == (second.visual_style_id or second.style_id)
728
+ )
729
+ return len(comparisons) >= 2 and sum(comparisons) / len(comparisons) >= 0.8
730
+
731
+
732
+ def _is_density_boundary(
733
+ occupied: dict[str, CellSnapshot],
734
+ positions: dict[str, tuple[int, int]],
735
+ rect: _Rect,
736
+ boundary: int,
737
+ ) -> bool:
738
+ left = _Rect(rect.min_column, rect.min_row, boundary, rect.max_row)
739
+ right = _Rect(boundary + 1, rect.min_row, rect.max_column, rect.max_row)
740
+ left_width = left.max_column - left.min_column + 1
741
+ right_width = right.max_column - right.min_column + 1
742
+ if min(left_width, right_width) != 1 or max(left_width, right_width) < 2:
743
+ return False
744
+ left_cells = [
745
+ occupied[coordinate]
746
+ for coordinate, position in positions.items()
747
+ if _contains(left, *position)
748
+ ]
749
+ right_cells = [
750
+ occupied[coordinate]
751
+ for coordinate, position in positions.items()
752
+ if _contains(right, *position)
753
+ ]
754
+ left_density = len(left_cells) / left.area
755
+ right_density = len(right_cells) / right.area
756
+ dense, sparse = (
757
+ (left_density, right_cells)
758
+ if left_density >= right_density
759
+ else (right_density, left_cells)
760
+ )
761
+ sparse_density = min(left_density, right_density)
762
+ return (
763
+ dense >= 0.75
764
+ and sparse_density <= 0.5
765
+ and any(len(cell.display_value.strip()) >= 20 for cell in sparse)
766
+ )
767
+
768
+
769
+ def _is_row_density_boundary(
770
+ occupied: dict[str, CellSnapshot],
771
+ positions: dict[str, tuple[int, int]],
772
+ rect: _Rect,
773
+ boundary: int,
774
+ ) -> bool:
775
+ top = _Rect(rect.min_column, rect.min_row, rect.max_column, boundary)
776
+ bottom = _Rect(rect.min_column, boundary + 1, rect.max_column, rect.max_row)
777
+ top_height = top.max_row - top.min_row + 1
778
+ bottom_height = bottom.max_row - bottom.min_row + 1
779
+ if min(top_height, bottom_height) != 1 or max(top_height, bottom_height) < 2:
780
+ return False
781
+ top_cells = [
782
+ occupied[coordinate]
783
+ for coordinate, position in positions.items()
784
+ if _contains(top, *position)
785
+ ]
786
+ bottom_cells = [
787
+ occupied[coordinate]
788
+ for coordinate, position in positions.items()
789
+ if _contains(bottom, *position)
790
+ ]
791
+ top_density = len(top_cells) / top.area
792
+ bottom_density = len(bottom_cells) / bottom.area
793
+ dense, sparse = (
794
+ (top_density, bottom_cells)
795
+ if top_density >= bottom_density
796
+ else (bottom_density, top_cells)
797
+ )
798
+ sparse_density = min(top_density, bottom_density)
799
+ return (
800
+ dense >= 0.75
801
+ and sparse_density <= 0.5
802
+ and any(len(cell.display_value.strip()) >= 20 for cell in sparse)
803
+ )
804
+
805
+
806
+ def _formula_boundaries(
807
+ occupied: dict[str, CellSnapshot],
808
+ positions: dict[str, tuple[int, int]],
809
+ rect: _Rect,
810
+ ) -> tuple[set[int], set[int]]:
811
+ columns: set[int] = set()
812
+ rows: set[int] = set()
813
+ for coordinate, cell in occupied.items():
814
+ if cell.formula is None:
815
+ continue
816
+ row, column = positions[coordinate]
817
+ if not _contains(rect, row, column):
818
+ continue
819
+ for ref_column, ref_row in _local_formula_refs(cell.formula):
820
+ if _contains(rect, ref_row, ref_column):
821
+ columns.update(range(min(column, ref_column), max(column, ref_column)))
822
+ rows.update(range(min(row, ref_row), max(row, ref_row)))
823
+ return columns, rows
824
+
825
+
826
+ def _local_formula_refs(formula: str) -> list[tuple[int, int]]:
827
+ try:
828
+ range_tokens = (
829
+ token.value
830
+ for token in Tokenizer(formula).items
831
+ if token.type == "OPERAND" and token.subtype == "RANGE" and "!" not in token.value
832
+ )
833
+ except Exception:
834
+ return []
835
+ return [
836
+ (column_index_from_string(column_name), int(row_number))
837
+ for value in range_tokens
838
+ for column_name, row_number in _FORMULA_CELL_REF.findall(value.upper())
839
+ ]
840
+
841
+
842
+ def _merged_range_crosses_column(sheet: SheetSnapshot, rect: _Rect, boundary: int) -> bool:
843
+ return any(
844
+ merged.min_column <= boundary < merged.max_column
845
+ and not (merged.max_row < rect.min_row or merged.min_row > rect.max_row)
846
+ for merged in _merged_rects(sheet)
847
+ )
848
+
849
+
850
+ def _merged_range_crosses_row(sheet: SheetSnapshot, rect: _Rect, boundary: int) -> bool:
851
+ return any(
852
+ merged.min_row <= boundary < merged.max_row
853
+ and not (merged.max_column < rect.min_column or merged.min_column > rect.max_column)
854
+ for merged in _merged_rects(sheet)
855
+ )
856
+
857
+
858
+ def _merged_rects(sheet: SheetSnapshot) -> list[_Rect]:
859
+ rectangles = []
860
+ for value in sheet.merged_ranges:
861
+ try:
862
+ rectangles.append(_rect_from_range(value.replace("$", "")))
863
+ except ValueError:
864
+ continue
865
+ return rectangles
866
+
867
+
868
+ def _anchor_reason(kind: str) -> str:
869
+ return "native_table_anchor" if kind == "excel_table" else "defined_name_anchor"
870
+
871
+
872
+ def _rect_from_range(value: str) -> _Rect:
873
+ boundaries = range_boundaries(value)
874
+ if any(boundary is None for boundary in boundaries):
875
+ raise ValueError("region ranges must have finite row and column bounds")
876
+ min_column, min_row, max_column, max_row = boundaries
877
+ return _Rect(min_column, min_row, max_column, max_row)
878
+
879
+
880
+ def _contains(rect: _Rect, row: int, column: int) -> bool:
881
+ return rect.min_row <= row <= rect.max_row and rect.min_column <= column <= rect.max_column
882
+
883
+
884
+ def _rect_inside(inner: _Rect, outer: _Rect) -> bool:
885
+ return (
886
+ outer.min_column <= inner.min_column <= inner.max_column <= outer.max_column
887
+ and outer.min_row <= inner.min_row <= inner.max_row <= outer.max_row
888
+ )
889
+
890
+
891
+ def _rectangles_overlap(left: _Rect, right: _Rect) -> bool:
892
+ return not (
893
+ left.max_column < right.min_column
894
+ or right.max_column < left.min_column
895
+ or left.max_row < right.min_row
896
+ or right.max_row < left.min_row
897
+ )
898
+
899
+
900
+ def _pairwise_non_overlapping(rectangles: list[_Rect]) -> bool:
901
+ return all(
902
+ not _rectangles_overlap(left, right)
903
+ for index, left in enumerate(rectangles)
904
+ for right in rectangles[index + 1 :]
905
+ )
906
+
907
+
908
+ def _is_occupied(cell: CellSnapshot) -> bool:
909
+ return any(
910
+ (
911
+ cell.raw_value is not None,
912
+ cell.formula is not None,
913
+ cell.comment is not None,
914
+ cell.hyperlink is not None,
915
+ cell.merge_anchor is not None,
916
+ )
917
+ )
918
+
919
+
920
+ def _consecutive_groups(values: Iterable[int]) -> list[tuple[int, int]]:
921
+ ordered = sorted(set(values))
922
+ if not ordered:
923
+ return []
924
+ groups: list[tuple[int, int]] = []
925
+ start = previous = ordered[0]
926
+ for value in ordered[1:]:
927
+ if value != previous + 1:
928
+ groups.append((start, previous))
929
+ start = value
930
+ previous = value
931
+ groups.append((start, previous))
932
+ return groups