langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,474 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
import re
|
|
6
|
+
import warnings as python_warnings
|
|
7
|
+
from dataclasses import asdict
|
|
8
|
+
from datetime import date, datetime, time
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any, Protocol
|
|
11
|
+
|
|
12
|
+
from openpyxl import load_workbook
|
|
13
|
+
from openpyxl.cell.cell import MergedCell
|
|
14
|
+
from openpyxl.utils import get_column_letter, range_boundaries
|
|
15
|
+
|
|
16
|
+
from langparse.workbooks.reference_types import (
|
|
17
|
+
DefinedNameFact,
|
|
18
|
+
ExcelTableFact,
|
|
19
|
+
WorkbookReferenceFacts,
|
|
20
|
+
)
|
|
21
|
+
from langparse.workbooks.references import extract_reference_facts
|
|
22
|
+
from langparse.workbooks.types import (
|
|
23
|
+
CellSnapshot,
|
|
24
|
+
RegionAnchor,
|
|
25
|
+
SheetSnapshot,
|
|
26
|
+
SourceRef,
|
|
27
|
+
WorkbookSnapshot,
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class WorkbookAdapter(Protocol):
|
|
32
|
+
def snapshot(self, path: str | Path) -> WorkbookSnapshot: ...
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class OOXMLWorkbookAdapter:
|
|
36
|
+
"""Extract OOXML workbook facts without interpreting table semantics."""
|
|
37
|
+
|
|
38
|
+
def snapshot(self, path: str | Path) -> WorkbookSnapshot:
|
|
39
|
+
workbook_path = Path(path)
|
|
40
|
+
keep_vba = workbook_path.suffix.lower() == ".xlsm"
|
|
41
|
+
# File-like input intentionally bypasses openpyxl's extension gate.
|
|
42
|
+
# LangParse routes by content, so a valid OOXML workbook renamed to
|
|
43
|
+
# ``.csv`` must still be readable as a workbook.
|
|
44
|
+
formula_stream = workbook_path.open("rb")
|
|
45
|
+
value_stream = workbook_path.open("rb")
|
|
46
|
+
warnings: list[str] = []
|
|
47
|
+
try:
|
|
48
|
+
with python_warnings.catch_warnings(record=True) as emitted:
|
|
49
|
+
python_warnings.simplefilter("always")
|
|
50
|
+
formula_book = load_workbook(
|
|
51
|
+
formula_stream,
|
|
52
|
+
data_only=False,
|
|
53
|
+
keep_vba=keep_vba,
|
|
54
|
+
read_only=False,
|
|
55
|
+
)
|
|
56
|
+
value_book = load_workbook(
|
|
57
|
+
value_stream,
|
|
58
|
+
data_only=True,
|
|
59
|
+
keep_vba=keep_vba,
|
|
60
|
+
read_only=False,
|
|
61
|
+
)
|
|
62
|
+
read_warnings = list(dict.fromkeys(str(item.message) for item in emitted))
|
|
63
|
+
warnings.extend(read_warnings)
|
|
64
|
+
drawing_read_errors = [
|
|
65
|
+
message for message in read_warnings if _drawing_read_loss(message)
|
|
66
|
+
]
|
|
67
|
+
reference_facts = WorkbookReferenceFacts()
|
|
68
|
+
for owner, scope in [
|
|
69
|
+
(formula_book, None),
|
|
70
|
+
*[(s, s.title) for s in formula_book.worksheets],
|
|
71
|
+
]:
|
|
72
|
+
reference_facts.defined_names.extend(
|
|
73
|
+
DefinedNameFact(name, definition.attr_text or "", scope)
|
|
74
|
+
for name, definition in sorted(owner.defined_names.items())
|
|
75
|
+
)
|
|
76
|
+
for sheet in formula_book.worksheets:
|
|
77
|
+
reference_facts.tables.extend(
|
|
78
|
+
ExcelTableFact(
|
|
79
|
+
table.name,
|
|
80
|
+
SourceRef(sheet.title, table.ref),
|
|
81
|
+
[column.name for column in table.tableColumns],
|
|
82
|
+
table.headerRowCount if table.headerRowCount is not None else 1,
|
|
83
|
+
table.totalsRowCount or 0,
|
|
84
|
+
)
|
|
85
|
+
for table in sorted(sheet.tables.values(), key=lambda t: t.name)
|
|
86
|
+
)
|
|
87
|
+
defined_name_anchors = _defined_name_region_anchors(
|
|
88
|
+
formula_book,
|
|
89
|
+
warnings,
|
|
90
|
+
scope="workbook",
|
|
91
|
+
)
|
|
92
|
+
local_defined_name_anchors = {
|
|
93
|
+
sheet.title: _defined_name_region_anchors(
|
|
94
|
+
sheet,
|
|
95
|
+
warnings,
|
|
96
|
+
scope="worksheet",
|
|
97
|
+
).get(sheet.title, [])
|
|
98
|
+
for sheet in formula_book.worksheets
|
|
99
|
+
}
|
|
100
|
+
local_defined_names = {
|
|
101
|
+
sheet.title: {name.casefold() for name in sheet.defined_names}
|
|
102
|
+
for sheet in formula_book.worksheets
|
|
103
|
+
}
|
|
104
|
+
sheets = [
|
|
105
|
+
self._snapshot_sheet(
|
|
106
|
+
formula_sheet,
|
|
107
|
+
value_book[formula_sheet.title],
|
|
108
|
+
index,
|
|
109
|
+
warnings,
|
|
110
|
+
_merge_defined_name_anchors(
|
|
111
|
+
defined_name_anchors.get(formula_sheet.title, []),
|
|
112
|
+
local_defined_name_anchors.get(formula_sheet.title, []),
|
|
113
|
+
local_defined_names.get(formula_sheet.title, set()),
|
|
114
|
+
),
|
|
115
|
+
)
|
|
116
|
+
for index, formula_sheet in enumerate(formula_book.worksheets)
|
|
117
|
+
]
|
|
118
|
+
finally:
|
|
119
|
+
if "formula_book" in locals():
|
|
120
|
+
formula_book.close()
|
|
121
|
+
if "value_book" in locals():
|
|
122
|
+
value_book.close()
|
|
123
|
+
formula_stream.close()
|
|
124
|
+
value_stream.close()
|
|
125
|
+
|
|
126
|
+
snapshot = WorkbookSnapshot(
|
|
127
|
+
source=str(workbook_path),
|
|
128
|
+
filename=workbook_path.name,
|
|
129
|
+
sheets=sheets,
|
|
130
|
+
reference_facts=reference_facts,
|
|
131
|
+
metadata={
|
|
132
|
+
"format": workbook_path.suffix.lower(),
|
|
133
|
+
"warnings": warnings,
|
|
134
|
+
"drawing_read_errors": drawing_read_errors,
|
|
135
|
+
},
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
extract_reference_facts(snapshot)
|
|
139
|
+
return snapshot
|
|
140
|
+
|
|
141
|
+
def _snapshot_sheet(
|
|
142
|
+
self,
|
|
143
|
+
formula_sheet: Any,
|
|
144
|
+
value_sheet: Any,
|
|
145
|
+
index: int,
|
|
146
|
+
warnings: list[str],
|
|
147
|
+
defined_name_anchors: list[RegionAnchor],
|
|
148
|
+
) -> SheetSnapshot:
|
|
149
|
+
hidden_rows = sorted(
|
|
150
|
+
row_index
|
|
151
|
+
for row_index, dimension in formula_sheet.row_dimensions.items()
|
|
152
|
+
if dimension.hidden
|
|
153
|
+
)
|
|
154
|
+
hidden_columns = sorted(
|
|
155
|
+
column
|
|
156
|
+
for column, dimension in formula_sheet.column_dimensions.items()
|
|
157
|
+
if dimension.hidden
|
|
158
|
+
)
|
|
159
|
+
row_heights = {
|
|
160
|
+
row_index: float(dimension.height)
|
|
161
|
+
for row_index, dimension in formula_sheet.row_dimensions.items()
|
|
162
|
+
if dimension.height is not None
|
|
163
|
+
}
|
|
164
|
+
column_widths = {
|
|
165
|
+
column: float(dimension.width)
|
|
166
|
+
for column, dimension in formula_sheet.column_dimensions.items()
|
|
167
|
+
if dimension.width is not None
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
cells: dict[str, CellSnapshot] = {}
|
|
171
|
+
for row in formula_sheet.iter_rows():
|
|
172
|
+
for cell in row:
|
|
173
|
+
if isinstance(cell, MergedCell) or not _should_capture(cell):
|
|
174
|
+
continue
|
|
175
|
+
cached_cell = value_sheet[cell.coordinate]
|
|
176
|
+
formula = cell.value if cell.data_type == "f" else None
|
|
177
|
+
cached_value = cached_cell.value if formula is not None else cell.value
|
|
178
|
+
cells[cell.coordinate] = CellSnapshot(
|
|
179
|
+
coordinate=cell.coordinate,
|
|
180
|
+
raw_value=cell.value,
|
|
181
|
+
display_value=_display_value(
|
|
182
|
+
cached_value if cached_value is not None else cell.value
|
|
183
|
+
),
|
|
184
|
+
formula=formula,
|
|
185
|
+
cached_value=cached_value,
|
|
186
|
+
data_type=str(cell.data_type or ""),
|
|
187
|
+
number_format=cell.number_format or "General",
|
|
188
|
+
style_id=_style_fingerprint(cell),
|
|
189
|
+
visual_style_id=_style_fingerprint(cell, visual_only=True),
|
|
190
|
+
hyperlink=_hyperlink_value(cell.hyperlink),
|
|
191
|
+
comment=cell.comment.text if cell.comment is not None else None,
|
|
192
|
+
hidden=cell.row in hidden_rows or cell.column_letter in hidden_columns,
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
merged_ranges = sorted(str(cell_range) for cell_range in formula_sheet.merged_cells.ranges)
|
|
196
|
+
for merged_range in formula_sheet.merged_cells.ranges:
|
|
197
|
+
min_col, min_row, max_col, max_row = range_boundaries(str(merged_range))
|
|
198
|
+
anchor = f"{get_column_letter(min_col)}{min_row}"
|
|
199
|
+
anchor_cell = formula_sheet[anchor]
|
|
200
|
+
anchor_snapshot = cells.setdefault(
|
|
201
|
+
anchor,
|
|
202
|
+
CellSnapshot(
|
|
203
|
+
coordinate=anchor,
|
|
204
|
+
raw_value=anchor_cell.value,
|
|
205
|
+
display_value=_display_value(anchor_cell.value),
|
|
206
|
+
data_type=str(anchor_cell.data_type or ""),
|
|
207
|
+
number_format=anchor_cell.number_format or "General",
|
|
208
|
+
style_id=_style_fingerprint(anchor_cell),
|
|
209
|
+
visual_style_id=_style_fingerprint(anchor_cell, visual_only=True),
|
|
210
|
+
),
|
|
211
|
+
)
|
|
212
|
+
anchor_snapshot.rowspan = max_row - min_row + 1
|
|
213
|
+
anchor_snapshot.colspan = max_col - min_col + 1
|
|
214
|
+
for row_index in range(min_row, max_row + 1):
|
|
215
|
+
for column_index in range(min_col, max_col + 1):
|
|
216
|
+
coordinate = f"{get_column_letter(column_index)}{row_index}"
|
|
217
|
+
if coordinate == anchor:
|
|
218
|
+
continue
|
|
219
|
+
cells.setdefault(
|
|
220
|
+
coordinate,
|
|
221
|
+
CellSnapshot(coordinate=coordinate, merge_anchor=anchor),
|
|
222
|
+
).merge_anchor = anchor
|
|
223
|
+
|
|
224
|
+
from langparse.workbooks.objects import extract_objects
|
|
225
|
+
|
|
226
|
+
objects = [asdict(obj) for obj in extract_objects(formula_sheet)]
|
|
227
|
+
warnings.extend(note for obj in objects for note in obj["diagnostics"])
|
|
228
|
+
|
|
229
|
+
region_anchors = [
|
|
230
|
+
RegionAnchor(
|
|
231
|
+
kind="excel_table",
|
|
232
|
+
source_ref=SourceRef(
|
|
233
|
+
sheet_name=formula_sheet.title,
|
|
234
|
+
range=_normalize_anchor_range(table.ref),
|
|
235
|
+
),
|
|
236
|
+
name=table.name,
|
|
237
|
+
scope="worksheet",
|
|
238
|
+
)
|
|
239
|
+
for table in sorted(formula_sheet.tables.values(), key=lambda item: item.name)
|
|
240
|
+
]
|
|
241
|
+
region_anchors.extend(defined_name_anchors)
|
|
242
|
+
region_anchors.sort(
|
|
243
|
+
key=lambda item: (
|
|
244
|
+
{"excel_table": 0, "defined_name": 1}.get(item.kind, 99),
|
|
245
|
+
item.source_ref.range,
|
|
246
|
+
item.name or "",
|
|
247
|
+
)
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
return SheetSnapshot(
|
|
251
|
+
name=formula_sheet.title,
|
|
252
|
+
index=index,
|
|
253
|
+
visibility=formula_sheet.sheet_state,
|
|
254
|
+
used_range=_used_range(cells, region_anchors),
|
|
255
|
+
print_area=_print_areas(formula_sheet.print_area),
|
|
256
|
+
row_heights=row_heights,
|
|
257
|
+
column_widths=column_widths,
|
|
258
|
+
hidden_rows=hidden_rows,
|
|
259
|
+
hidden_columns=hidden_columns,
|
|
260
|
+
merged_ranges=merged_ranges,
|
|
261
|
+
cells=dict(sorted(cells.items(), key=lambda item: _coordinate_sort_key(item[0]))),
|
|
262
|
+
objects=objects,
|
|
263
|
+
region_anchors=region_anchors,
|
|
264
|
+
)
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def _defined_name_region_anchors(
|
|
268
|
+
owner: Any,
|
|
269
|
+
warnings: list[str],
|
|
270
|
+
*,
|
|
271
|
+
scope: str,
|
|
272
|
+
) -> dict[str, list[RegionAnchor]]:
|
|
273
|
+
anchors: dict[str, list[RegionAnchor]] = {}
|
|
274
|
+
for name, definition in sorted(owner.defined_names.items()):
|
|
275
|
+
try:
|
|
276
|
+
destinations = list(definition.destinations)
|
|
277
|
+
except (AttributeError, TypeError, ValueError):
|
|
278
|
+
warnings.append("defined_name_anchor_unsupported")
|
|
279
|
+
continue
|
|
280
|
+
for sheet_name, target in destinations:
|
|
281
|
+
try:
|
|
282
|
+
normalized = _normalize_anchor_range(target)
|
|
283
|
+
except ValueError:
|
|
284
|
+
warnings.append("defined_name_anchor_unsupported")
|
|
285
|
+
continue
|
|
286
|
+
anchors.setdefault(sheet_name, []).append(
|
|
287
|
+
RegionAnchor(
|
|
288
|
+
kind="defined_name",
|
|
289
|
+
source_ref=SourceRef(sheet_name=sheet_name, range=normalized),
|
|
290
|
+
name=name,
|
|
291
|
+
scope=scope,
|
|
292
|
+
)
|
|
293
|
+
)
|
|
294
|
+
return anchors
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def _merge_defined_name_anchors(
|
|
298
|
+
workbook_anchors: list[RegionAnchor],
|
|
299
|
+
worksheet_anchors: list[RegionAnchor],
|
|
300
|
+
worksheet_names: set[str],
|
|
301
|
+
) -> list[RegionAnchor]:
|
|
302
|
+
return [
|
|
303
|
+
*[
|
|
304
|
+
anchor
|
|
305
|
+
for anchor in workbook_anchors
|
|
306
|
+
if anchor.name is None or anchor.name.casefold() not in worksheet_names
|
|
307
|
+
],
|
|
308
|
+
*worksheet_anchors,
|
|
309
|
+
]
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def _normalize_anchor_range(value: str) -> str:
|
|
313
|
+
normalized = value.replace("$", "")
|
|
314
|
+
if "," in normalized or "!" in normalized:
|
|
315
|
+
raise ValueError("region anchor must be a single local range")
|
|
316
|
+
boundaries = range_boundaries(normalized)
|
|
317
|
+
if any(boundary is None for boundary in boundaries):
|
|
318
|
+
raise ValueError("region anchor must have finite row and column bounds")
|
|
319
|
+
return normalized
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def _should_capture(cell: Any) -> bool:
|
|
323
|
+
return any(
|
|
324
|
+
(
|
|
325
|
+
cell.value is not None,
|
|
326
|
+
cell.has_style,
|
|
327
|
+
cell.comment is not None,
|
|
328
|
+
cell.hyperlink is not None,
|
|
329
|
+
)
|
|
330
|
+
)
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def _display_value(value: Any) -> str:
|
|
334
|
+
if value is None:
|
|
335
|
+
return ""
|
|
336
|
+
if isinstance(value, (datetime, date, time)):
|
|
337
|
+
return value.isoformat()
|
|
338
|
+
return str(value)
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def _hyperlink_value(hyperlink: Any) -> str | None:
|
|
342
|
+
if hyperlink is None:
|
|
343
|
+
return None
|
|
344
|
+
return hyperlink.target or hyperlink.location
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def _color_payload(color: Any) -> dict[str, Any] | None:
|
|
348
|
+
if color is None:
|
|
349
|
+
return None
|
|
350
|
+
return {
|
|
351
|
+
"type": color.type,
|
|
352
|
+
"rgb": color.rgb if color.type == "rgb" else None,
|
|
353
|
+
"indexed": color.indexed if color.type == "indexed" else None,
|
|
354
|
+
"theme": color.theme if color.type == "theme" else None,
|
|
355
|
+
"tint": color.tint,
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
|
|
359
|
+
def _side_payload(side: Any) -> dict[str, Any]:
|
|
360
|
+
return {"style": side.style, "color": _color_payload(side.color)}
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
def _style_fingerprint(cell: Any, *, visual_only: bool = False) -> str:
|
|
364
|
+
if not cell.has_style and not visual_only:
|
|
365
|
+
return ""
|
|
366
|
+
payload = {
|
|
367
|
+
"font": {
|
|
368
|
+
"name": cell.font.name,
|
|
369
|
+
"size": cell.font.sz,
|
|
370
|
+
"bold": cell.font.b,
|
|
371
|
+
"italic": cell.font.i,
|
|
372
|
+
"underline": cell.font.u,
|
|
373
|
+
"strike": cell.font.strike,
|
|
374
|
+
"color": _color_payload(cell.font.color),
|
|
375
|
+
},
|
|
376
|
+
"fill": {
|
|
377
|
+
"type": cell.fill.fill_type,
|
|
378
|
+
"foreground": _color_payload(cell.fill.fgColor),
|
|
379
|
+
"background": _color_payload(cell.fill.bgColor),
|
|
380
|
+
},
|
|
381
|
+
"border": {
|
|
382
|
+
"left": _side_payload(cell.border.left),
|
|
383
|
+
"right": _side_payload(cell.border.right),
|
|
384
|
+
"top": _side_payload(cell.border.top),
|
|
385
|
+
"bottom": _side_payload(cell.border.bottom),
|
|
386
|
+
},
|
|
387
|
+
"alignment": {
|
|
388
|
+
"horizontal": cell.alignment.horizontal,
|
|
389
|
+
"vertical": cell.alignment.vertical,
|
|
390
|
+
"wrap_text": cell.alignment.wrap_text,
|
|
391
|
+
"text_rotation": cell.alignment.text_rotation,
|
|
392
|
+
},
|
|
393
|
+
}
|
|
394
|
+
if not visual_only:
|
|
395
|
+
payload["number_format"] = cell.number_format
|
|
396
|
+
payload["protection"] = {
|
|
397
|
+
"locked": cell.protection.locked,
|
|
398
|
+
"hidden": cell.protection.hidden,
|
|
399
|
+
}
|
|
400
|
+
encoded = json.dumps(payload, ensure_ascii=False, sort_keys=True, default=str).encode("utf-8")
|
|
401
|
+
return hashlib.sha256(encoded).hexdigest()[:16]
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
def _coordinate_sort_key(coordinate: str) -> tuple[int, int]:
|
|
405
|
+
match = re.fullmatch(r"([A-Z]+)([0-9]+)", coordinate)
|
|
406
|
+
if match is None:
|
|
407
|
+
return (0, 0)
|
|
408
|
+
column = 0
|
|
409
|
+
for character in match.group(1):
|
|
410
|
+
column = column * 26 + ord(character) - ord("A") + 1
|
|
411
|
+
return (int(match.group(2)), column)
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def _used_range(
|
|
415
|
+
cells: dict[str, CellSnapshot],
|
|
416
|
+
region_anchors: list[RegionAnchor],
|
|
417
|
+
) -> str | None:
|
|
418
|
+
table_ranges = [
|
|
419
|
+
range_boundaries(anchor.source_ref.range)
|
|
420
|
+
for anchor in region_anchors
|
|
421
|
+
if anchor.kind == "excel_table"
|
|
422
|
+
]
|
|
423
|
+
if not cells and not table_ranges:
|
|
424
|
+
return None
|
|
425
|
+
coordinates = [_coordinate_sort_key(coordinate) for coordinate in cells]
|
|
426
|
+
rows = [
|
|
427
|
+
*[row for row, _ in coordinates],
|
|
428
|
+
*[row for _, min_row, _, max_row in table_ranges for row in (min_row, max_row)],
|
|
429
|
+
]
|
|
430
|
+
columns = [
|
|
431
|
+
*[column for _, column in coordinates],
|
|
432
|
+
*[
|
|
433
|
+
column
|
|
434
|
+
for min_column, _, max_column, _ in table_ranges
|
|
435
|
+
for column in (min_column, max_column)
|
|
436
|
+
],
|
|
437
|
+
]
|
|
438
|
+
return (
|
|
439
|
+
f"{get_column_letter(min(columns))}{min(rows)}:{get_column_letter(max(columns))}{max(rows)}"
|
|
440
|
+
)
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
def _print_areas(print_area: Any) -> list[str]:
|
|
444
|
+
if not print_area:
|
|
445
|
+
return []
|
|
446
|
+
parts = re.split(r",(?=(?:[^']*'[^']*')*[^']*$)", str(print_area))
|
|
447
|
+
return [re.sub(r"^'([^']+)'!", r"\1!", part.strip()) for part in parts]
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
def _object_anchor(obj: Any, sheet_name: str, warnings: list[str]) -> str | None:
|
|
451
|
+
anchor = getattr(obj, "anchor", None)
|
|
452
|
+
if isinstance(anchor, str):
|
|
453
|
+
return anchor
|
|
454
|
+
marker = getattr(anchor, "_from", None)
|
|
455
|
+
if marker is not None:
|
|
456
|
+
return f"{get_column_letter(marker.col + 1)}{marker.row + 1}"
|
|
457
|
+
warnings.append(f"Unable to resolve object anchor on sheet {sheet_name}")
|
|
458
|
+
return None
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def _chart_title(chart: Any) -> str | None:
|
|
462
|
+
title = getattr(chart, "title", None)
|
|
463
|
+
if title is None:
|
|
464
|
+
return None
|
|
465
|
+
return str(title)
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
def _drawing_read_loss(message: str) -> bool:
|
|
469
|
+
lowered = message.lower()
|
|
470
|
+
return (
|
|
471
|
+
"drawingml support is incomplete" in lowered
|
|
472
|
+
or "unable to read chart" in lowered
|
|
473
|
+
or ("image" in lowered and ("removed" in lowered or "dropped" in lowered))
|
|
474
|
+
)
|