langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,474 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import json
5
+ import re
6
+ import warnings as python_warnings
7
+ from dataclasses import asdict
8
+ from datetime import date, datetime, time
9
+ from pathlib import Path
10
+ from typing import Any, Protocol
11
+
12
+ from openpyxl import load_workbook
13
+ from openpyxl.cell.cell import MergedCell
14
+ from openpyxl.utils import get_column_letter, range_boundaries
15
+
16
+ from langparse.workbooks.reference_types import (
17
+ DefinedNameFact,
18
+ ExcelTableFact,
19
+ WorkbookReferenceFacts,
20
+ )
21
+ from langparse.workbooks.references import extract_reference_facts
22
+ from langparse.workbooks.types import (
23
+ CellSnapshot,
24
+ RegionAnchor,
25
+ SheetSnapshot,
26
+ SourceRef,
27
+ WorkbookSnapshot,
28
+ )
29
+
30
+
31
+ class WorkbookAdapter(Protocol):
32
+ def snapshot(self, path: str | Path) -> WorkbookSnapshot: ...
33
+
34
+
35
+ class OOXMLWorkbookAdapter:
36
+ """Extract OOXML workbook facts without interpreting table semantics."""
37
+
38
+ def snapshot(self, path: str | Path) -> WorkbookSnapshot:
39
+ workbook_path = Path(path)
40
+ keep_vba = workbook_path.suffix.lower() == ".xlsm"
41
+ # File-like input intentionally bypasses openpyxl's extension gate.
42
+ # LangParse routes by content, so a valid OOXML workbook renamed to
43
+ # ``.csv`` must still be readable as a workbook.
44
+ formula_stream = workbook_path.open("rb")
45
+ value_stream = workbook_path.open("rb")
46
+ warnings: list[str] = []
47
+ try:
48
+ with python_warnings.catch_warnings(record=True) as emitted:
49
+ python_warnings.simplefilter("always")
50
+ formula_book = load_workbook(
51
+ formula_stream,
52
+ data_only=False,
53
+ keep_vba=keep_vba,
54
+ read_only=False,
55
+ )
56
+ value_book = load_workbook(
57
+ value_stream,
58
+ data_only=True,
59
+ keep_vba=keep_vba,
60
+ read_only=False,
61
+ )
62
+ read_warnings = list(dict.fromkeys(str(item.message) for item in emitted))
63
+ warnings.extend(read_warnings)
64
+ drawing_read_errors = [
65
+ message for message in read_warnings if _drawing_read_loss(message)
66
+ ]
67
+ reference_facts = WorkbookReferenceFacts()
68
+ for owner, scope in [
69
+ (formula_book, None),
70
+ *[(s, s.title) for s in formula_book.worksheets],
71
+ ]:
72
+ reference_facts.defined_names.extend(
73
+ DefinedNameFact(name, definition.attr_text or "", scope)
74
+ for name, definition in sorted(owner.defined_names.items())
75
+ )
76
+ for sheet in formula_book.worksheets:
77
+ reference_facts.tables.extend(
78
+ ExcelTableFact(
79
+ table.name,
80
+ SourceRef(sheet.title, table.ref),
81
+ [column.name for column in table.tableColumns],
82
+ table.headerRowCount if table.headerRowCount is not None else 1,
83
+ table.totalsRowCount or 0,
84
+ )
85
+ for table in sorted(sheet.tables.values(), key=lambda t: t.name)
86
+ )
87
+ defined_name_anchors = _defined_name_region_anchors(
88
+ formula_book,
89
+ warnings,
90
+ scope="workbook",
91
+ )
92
+ local_defined_name_anchors = {
93
+ sheet.title: _defined_name_region_anchors(
94
+ sheet,
95
+ warnings,
96
+ scope="worksheet",
97
+ ).get(sheet.title, [])
98
+ for sheet in formula_book.worksheets
99
+ }
100
+ local_defined_names = {
101
+ sheet.title: {name.casefold() for name in sheet.defined_names}
102
+ for sheet in formula_book.worksheets
103
+ }
104
+ sheets = [
105
+ self._snapshot_sheet(
106
+ formula_sheet,
107
+ value_book[formula_sheet.title],
108
+ index,
109
+ warnings,
110
+ _merge_defined_name_anchors(
111
+ defined_name_anchors.get(formula_sheet.title, []),
112
+ local_defined_name_anchors.get(formula_sheet.title, []),
113
+ local_defined_names.get(formula_sheet.title, set()),
114
+ ),
115
+ )
116
+ for index, formula_sheet in enumerate(formula_book.worksheets)
117
+ ]
118
+ finally:
119
+ if "formula_book" in locals():
120
+ formula_book.close()
121
+ if "value_book" in locals():
122
+ value_book.close()
123
+ formula_stream.close()
124
+ value_stream.close()
125
+
126
+ snapshot = WorkbookSnapshot(
127
+ source=str(workbook_path),
128
+ filename=workbook_path.name,
129
+ sheets=sheets,
130
+ reference_facts=reference_facts,
131
+ metadata={
132
+ "format": workbook_path.suffix.lower(),
133
+ "warnings": warnings,
134
+ "drawing_read_errors": drawing_read_errors,
135
+ },
136
+ )
137
+
138
+ extract_reference_facts(snapshot)
139
+ return snapshot
140
+
141
+ def _snapshot_sheet(
142
+ self,
143
+ formula_sheet: Any,
144
+ value_sheet: Any,
145
+ index: int,
146
+ warnings: list[str],
147
+ defined_name_anchors: list[RegionAnchor],
148
+ ) -> SheetSnapshot:
149
+ hidden_rows = sorted(
150
+ row_index
151
+ for row_index, dimension in formula_sheet.row_dimensions.items()
152
+ if dimension.hidden
153
+ )
154
+ hidden_columns = sorted(
155
+ column
156
+ for column, dimension in formula_sheet.column_dimensions.items()
157
+ if dimension.hidden
158
+ )
159
+ row_heights = {
160
+ row_index: float(dimension.height)
161
+ for row_index, dimension in formula_sheet.row_dimensions.items()
162
+ if dimension.height is not None
163
+ }
164
+ column_widths = {
165
+ column: float(dimension.width)
166
+ for column, dimension in formula_sheet.column_dimensions.items()
167
+ if dimension.width is not None
168
+ }
169
+
170
+ cells: dict[str, CellSnapshot] = {}
171
+ for row in formula_sheet.iter_rows():
172
+ for cell in row:
173
+ if isinstance(cell, MergedCell) or not _should_capture(cell):
174
+ continue
175
+ cached_cell = value_sheet[cell.coordinate]
176
+ formula = cell.value if cell.data_type == "f" else None
177
+ cached_value = cached_cell.value if formula is not None else cell.value
178
+ cells[cell.coordinate] = CellSnapshot(
179
+ coordinate=cell.coordinate,
180
+ raw_value=cell.value,
181
+ display_value=_display_value(
182
+ cached_value if cached_value is not None else cell.value
183
+ ),
184
+ formula=formula,
185
+ cached_value=cached_value,
186
+ data_type=str(cell.data_type or ""),
187
+ number_format=cell.number_format or "General",
188
+ style_id=_style_fingerprint(cell),
189
+ visual_style_id=_style_fingerprint(cell, visual_only=True),
190
+ hyperlink=_hyperlink_value(cell.hyperlink),
191
+ comment=cell.comment.text if cell.comment is not None else None,
192
+ hidden=cell.row in hidden_rows or cell.column_letter in hidden_columns,
193
+ )
194
+
195
+ merged_ranges = sorted(str(cell_range) for cell_range in formula_sheet.merged_cells.ranges)
196
+ for merged_range in formula_sheet.merged_cells.ranges:
197
+ min_col, min_row, max_col, max_row = range_boundaries(str(merged_range))
198
+ anchor = f"{get_column_letter(min_col)}{min_row}"
199
+ anchor_cell = formula_sheet[anchor]
200
+ anchor_snapshot = cells.setdefault(
201
+ anchor,
202
+ CellSnapshot(
203
+ coordinate=anchor,
204
+ raw_value=anchor_cell.value,
205
+ display_value=_display_value(anchor_cell.value),
206
+ data_type=str(anchor_cell.data_type or ""),
207
+ number_format=anchor_cell.number_format or "General",
208
+ style_id=_style_fingerprint(anchor_cell),
209
+ visual_style_id=_style_fingerprint(anchor_cell, visual_only=True),
210
+ ),
211
+ )
212
+ anchor_snapshot.rowspan = max_row - min_row + 1
213
+ anchor_snapshot.colspan = max_col - min_col + 1
214
+ for row_index in range(min_row, max_row + 1):
215
+ for column_index in range(min_col, max_col + 1):
216
+ coordinate = f"{get_column_letter(column_index)}{row_index}"
217
+ if coordinate == anchor:
218
+ continue
219
+ cells.setdefault(
220
+ coordinate,
221
+ CellSnapshot(coordinate=coordinate, merge_anchor=anchor),
222
+ ).merge_anchor = anchor
223
+
224
+ from langparse.workbooks.objects import extract_objects
225
+
226
+ objects = [asdict(obj) for obj in extract_objects(formula_sheet)]
227
+ warnings.extend(note for obj in objects for note in obj["diagnostics"])
228
+
229
+ region_anchors = [
230
+ RegionAnchor(
231
+ kind="excel_table",
232
+ source_ref=SourceRef(
233
+ sheet_name=formula_sheet.title,
234
+ range=_normalize_anchor_range(table.ref),
235
+ ),
236
+ name=table.name,
237
+ scope="worksheet",
238
+ )
239
+ for table in sorted(formula_sheet.tables.values(), key=lambda item: item.name)
240
+ ]
241
+ region_anchors.extend(defined_name_anchors)
242
+ region_anchors.sort(
243
+ key=lambda item: (
244
+ {"excel_table": 0, "defined_name": 1}.get(item.kind, 99),
245
+ item.source_ref.range,
246
+ item.name or "",
247
+ )
248
+ )
249
+
250
+ return SheetSnapshot(
251
+ name=formula_sheet.title,
252
+ index=index,
253
+ visibility=formula_sheet.sheet_state,
254
+ used_range=_used_range(cells, region_anchors),
255
+ print_area=_print_areas(formula_sheet.print_area),
256
+ row_heights=row_heights,
257
+ column_widths=column_widths,
258
+ hidden_rows=hidden_rows,
259
+ hidden_columns=hidden_columns,
260
+ merged_ranges=merged_ranges,
261
+ cells=dict(sorted(cells.items(), key=lambda item: _coordinate_sort_key(item[0]))),
262
+ objects=objects,
263
+ region_anchors=region_anchors,
264
+ )
265
+
266
+
267
+ def _defined_name_region_anchors(
268
+ owner: Any,
269
+ warnings: list[str],
270
+ *,
271
+ scope: str,
272
+ ) -> dict[str, list[RegionAnchor]]:
273
+ anchors: dict[str, list[RegionAnchor]] = {}
274
+ for name, definition in sorted(owner.defined_names.items()):
275
+ try:
276
+ destinations = list(definition.destinations)
277
+ except (AttributeError, TypeError, ValueError):
278
+ warnings.append("defined_name_anchor_unsupported")
279
+ continue
280
+ for sheet_name, target in destinations:
281
+ try:
282
+ normalized = _normalize_anchor_range(target)
283
+ except ValueError:
284
+ warnings.append("defined_name_anchor_unsupported")
285
+ continue
286
+ anchors.setdefault(sheet_name, []).append(
287
+ RegionAnchor(
288
+ kind="defined_name",
289
+ source_ref=SourceRef(sheet_name=sheet_name, range=normalized),
290
+ name=name,
291
+ scope=scope,
292
+ )
293
+ )
294
+ return anchors
295
+
296
+
297
+ def _merge_defined_name_anchors(
298
+ workbook_anchors: list[RegionAnchor],
299
+ worksheet_anchors: list[RegionAnchor],
300
+ worksheet_names: set[str],
301
+ ) -> list[RegionAnchor]:
302
+ return [
303
+ *[
304
+ anchor
305
+ for anchor in workbook_anchors
306
+ if anchor.name is None or anchor.name.casefold() not in worksheet_names
307
+ ],
308
+ *worksheet_anchors,
309
+ ]
310
+
311
+
312
+ def _normalize_anchor_range(value: str) -> str:
313
+ normalized = value.replace("$", "")
314
+ if "," in normalized or "!" in normalized:
315
+ raise ValueError("region anchor must be a single local range")
316
+ boundaries = range_boundaries(normalized)
317
+ if any(boundary is None for boundary in boundaries):
318
+ raise ValueError("region anchor must have finite row and column bounds")
319
+ return normalized
320
+
321
+
322
+ def _should_capture(cell: Any) -> bool:
323
+ return any(
324
+ (
325
+ cell.value is not None,
326
+ cell.has_style,
327
+ cell.comment is not None,
328
+ cell.hyperlink is not None,
329
+ )
330
+ )
331
+
332
+
333
+ def _display_value(value: Any) -> str:
334
+ if value is None:
335
+ return ""
336
+ if isinstance(value, (datetime, date, time)):
337
+ return value.isoformat()
338
+ return str(value)
339
+
340
+
341
+ def _hyperlink_value(hyperlink: Any) -> str | None:
342
+ if hyperlink is None:
343
+ return None
344
+ return hyperlink.target or hyperlink.location
345
+
346
+
347
+ def _color_payload(color: Any) -> dict[str, Any] | None:
348
+ if color is None:
349
+ return None
350
+ return {
351
+ "type": color.type,
352
+ "rgb": color.rgb if color.type == "rgb" else None,
353
+ "indexed": color.indexed if color.type == "indexed" else None,
354
+ "theme": color.theme if color.type == "theme" else None,
355
+ "tint": color.tint,
356
+ }
357
+
358
+
359
+ def _side_payload(side: Any) -> dict[str, Any]:
360
+ return {"style": side.style, "color": _color_payload(side.color)}
361
+
362
+
363
+ def _style_fingerprint(cell: Any, *, visual_only: bool = False) -> str:
364
+ if not cell.has_style and not visual_only:
365
+ return ""
366
+ payload = {
367
+ "font": {
368
+ "name": cell.font.name,
369
+ "size": cell.font.sz,
370
+ "bold": cell.font.b,
371
+ "italic": cell.font.i,
372
+ "underline": cell.font.u,
373
+ "strike": cell.font.strike,
374
+ "color": _color_payload(cell.font.color),
375
+ },
376
+ "fill": {
377
+ "type": cell.fill.fill_type,
378
+ "foreground": _color_payload(cell.fill.fgColor),
379
+ "background": _color_payload(cell.fill.bgColor),
380
+ },
381
+ "border": {
382
+ "left": _side_payload(cell.border.left),
383
+ "right": _side_payload(cell.border.right),
384
+ "top": _side_payload(cell.border.top),
385
+ "bottom": _side_payload(cell.border.bottom),
386
+ },
387
+ "alignment": {
388
+ "horizontal": cell.alignment.horizontal,
389
+ "vertical": cell.alignment.vertical,
390
+ "wrap_text": cell.alignment.wrap_text,
391
+ "text_rotation": cell.alignment.text_rotation,
392
+ },
393
+ }
394
+ if not visual_only:
395
+ payload["number_format"] = cell.number_format
396
+ payload["protection"] = {
397
+ "locked": cell.protection.locked,
398
+ "hidden": cell.protection.hidden,
399
+ }
400
+ encoded = json.dumps(payload, ensure_ascii=False, sort_keys=True, default=str).encode("utf-8")
401
+ return hashlib.sha256(encoded).hexdigest()[:16]
402
+
403
+
404
+ def _coordinate_sort_key(coordinate: str) -> tuple[int, int]:
405
+ match = re.fullmatch(r"([A-Z]+)([0-9]+)", coordinate)
406
+ if match is None:
407
+ return (0, 0)
408
+ column = 0
409
+ for character in match.group(1):
410
+ column = column * 26 + ord(character) - ord("A") + 1
411
+ return (int(match.group(2)), column)
412
+
413
+
414
+ def _used_range(
415
+ cells: dict[str, CellSnapshot],
416
+ region_anchors: list[RegionAnchor],
417
+ ) -> str | None:
418
+ table_ranges = [
419
+ range_boundaries(anchor.source_ref.range)
420
+ for anchor in region_anchors
421
+ if anchor.kind == "excel_table"
422
+ ]
423
+ if not cells and not table_ranges:
424
+ return None
425
+ coordinates = [_coordinate_sort_key(coordinate) for coordinate in cells]
426
+ rows = [
427
+ *[row for row, _ in coordinates],
428
+ *[row for _, min_row, _, max_row in table_ranges for row in (min_row, max_row)],
429
+ ]
430
+ columns = [
431
+ *[column for _, column in coordinates],
432
+ *[
433
+ column
434
+ for min_column, _, max_column, _ in table_ranges
435
+ for column in (min_column, max_column)
436
+ ],
437
+ ]
438
+ return (
439
+ f"{get_column_letter(min(columns))}{min(rows)}:{get_column_letter(max(columns))}{max(rows)}"
440
+ )
441
+
442
+
443
+ def _print_areas(print_area: Any) -> list[str]:
444
+ if not print_area:
445
+ return []
446
+ parts = re.split(r",(?=(?:[^']*'[^']*')*[^']*$)", str(print_area))
447
+ return [re.sub(r"^'([^']+)'!", r"\1!", part.strip()) for part in parts]
448
+
449
+
450
+ def _object_anchor(obj: Any, sheet_name: str, warnings: list[str]) -> str | None:
451
+ anchor = getattr(obj, "anchor", None)
452
+ if isinstance(anchor, str):
453
+ return anchor
454
+ marker = getattr(anchor, "_from", None)
455
+ if marker is not None:
456
+ return f"{get_column_letter(marker.col + 1)}{marker.row + 1}"
457
+ warnings.append(f"Unable to resolve object anchor on sheet {sheet_name}")
458
+ return None
459
+
460
+
461
+ def _chart_title(chart: Any) -> str | None:
462
+ title = getattr(chart, "title", None)
463
+ if title is None:
464
+ return None
465
+ return str(title)
466
+
467
+
468
+ def _drawing_read_loss(message: str) -> bool:
469
+ lowered = message.lower()
470
+ return (
471
+ "drawingml support is incomplete" in lowered
472
+ or "unable to read chart" in lowered
473
+ or ("image" in lowered and ("removed" in lowered or "dropped" in lowered))
474
+ )