langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,209 @@
1
+ from __future__ import annotations
2
+
3
+ from openpyxl.utils import get_column_letter, range_boundaries
4
+
5
+ from langparse.workbooks.classification import BlockClassification
6
+ from langparse.workbooks.types import (
7
+ CandidateRegion,
8
+ FormBlock,
9
+ FormField,
10
+ MatrixBlock,
11
+ MatrixHeader,
12
+ SheetSnapshot,
13
+ SourceRef,
14
+ TextBlock,
15
+ TextLine,
16
+ stable_id,
17
+ )
18
+
19
+
20
+ def interpret_form_block(
21
+ sheet: SheetSnapshot,
22
+ candidate: CandidateRegion,
23
+ classification: BlockClassification,
24
+ ) -> FormBlock:
25
+ """Interpret adjacent label/value pairs without inventing missing fields."""
26
+
27
+ min_col, min_row, max_col, max_row = range_boundaries(candidate.source_ref.range)
28
+ title = ""
29
+ fields: list[FormField] = []
30
+ free_text: list[TextLine] = []
31
+ for row_number in range(min_row, max_row + 1):
32
+ entries = _row_entries(sheet, row_number, min_col, max_col)
33
+ if row_number == min_row and len(entries) == 1:
34
+ title = entries[0][1]
35
+ continue
36
+ pairs = _adjacent_pairs(entries)
37
+ if pairs is not None:
38
+ for (label_coordinate, label), (value_coordinate, value) in pairs:
39
+ label_ref = SourceRef(sheet_name=sheet.name, range=label_coordinate)
40
+ value_ref = SourceRef(sheet_name=sheet.name, range=value_coordinate)
41
+ fields.append(
42
+ FormField(
43
+ field_id=stable_id(
44
+ "field",
45
+ candidate.source_ref.key,
46
+ label_ref.key,
47
+ value_ref.key,
48
+ ),
49
+ label=label,
50
+ value=value,
51
+ label_source_refs=[label_ref],
52
+ value_source_refs=[value_ref],
53
+ confidence=classification.confidence,
54
+ )
55
+ )
56
+ continue
57
+ if entries:
58
+ free_text.append(
59
+ TextLine(
60
+ text=" ".join(value for _, value in entries),
61
+ source_refs=[
62
+ SourceRef(sheet_name=sheet.name, range=coordinate)
63
+ for coordinate, _ in entries
64
+ ],
65
+ )
66
+ )
67
+ return FormBlock(
68
+ form_id=stable_id("form", candidate.source_ref.key),
69
+ title=title,
70
+ fields=fields,
71
+ free_text=free_text,
72
+ source_refs=[candidate.source_ref],
73
+ confidence=classification.confidence,
74
+ diagnostics=[{"reason_codes": list(classification.reason_codes)}],
75
+ )
76
+
77
+
78
+ def interpret_matrix_block(
79
+ sheet: SheetSnapshot,
80
+ candidate: CandidateRegion,
81
+ classification: BlockClassification,
82
+ ) -> MatrixBlock:
83
+ """Preserve a two-axis matrix and its physical value grid."""
84
+
85
+ min_col, min_row, max_col, max_row = range_boundaries(candidate.source_ref.range)
86
+ if max_col - min_col < 2 or max_row - min_row < 2:
87
+ raise ValueError("matrix requires at least two row and column dimensions")
88
+
89
+ title = _display_value(sheet, min_row, min_col)
90
+ column_headers = [
91
+ MatrixHeader(
92
+ value=_display_value(sheet, min_row, column),
93
+ source_refs=[_source_ref(sheet, min_row, column)],
94
+ )
95
+ for column in range(min_col + 1, max_col + 1)
96
+ ]
97
+ row_headers = [
98
+ MatrixHeader(
99
+ value=_display_value(sheet, row, min_col),
100
+ source_refs=[_source_ref(sheet, row, min_col)],
101
+ )
102
+ for row in range(min_row + 1, max_row + 1)
103
+ ]
104
+ if any(not header.value for header in [*column_headers, *row_headers]):
105
+ raise ValueError("matrix axes must be complete")
106
+
107
+ values = []
108
+ value_source_refs = []
109
+ for row in range(min_row + 1, max_row + 1):
110
+ value_row = []
111
+ ref_row = []
112
+ for column in range(min_col + 1, max_col + 1):
113
+ value_row.append(_display_value(sheet, row, column))
114
+ coordinate = f"{get_column_letter(column)}{row}"
115
+ ref_row.append(_source_ref(sheet, row, column) if coordinate in sheet.cells else None)
116
+ values.append(value_row)
117
+ value_source_refs.append(ref_row)
118
+
119
+ return MatrixBlock(
120
+ matrix_id=stable_id("matrix", candidate.source_ref.key),
121
+ title=title,
122
+ row_headers=row_headers,
123
+ column_headers=column_headers,
124
+ values=values,
125
+ source_refs=[candidate.source_ref],
126
+ value_source_refs=value_source_refs,
127
+ confidence=classification.confidence,
128
+ diagnostics=[{"reason_codes": list(classification.reason_codes)}],
129
+ )
130
+
131
+
132
+ def interpret_text_block(
133
+ sheet: SheetSnapshot,
134
+ candidate: CandidateRegion,
135
+ classification: BlockClassification,
136
+ ) -> TextBlock:
137
+ """Render source-ordered display text without merged-cell duplication."""
138
+
139
+ min_col, min_row, max_col, max_row = range_boundaries(candidate.source_ref.range)
140
+ lines = []
141
+ for row_number in range(min_row, max_row + 1):
142
+ entries = _row_entries(sheet, row_number, min_col, max_col)
143
+ if entries:
144
+ lines.append(
145
+ TextLine(
146
+ text=" ".join(value for _, value in entries),
147
+ source_refs=[
148
+ SourceRef(sheet_name=sheet.name, range=coordinate)
149
+ for coordinate, _ in entries
150
+ ],
151
+ )
152
+ )
153
+ return TextBlock(
154
+ text_id=stable_id("text", candidate.source_ref.key),
155
+ lines=lines,
156
+ source_refs=[candidate.source_ref],
157
+ confidence=classification.confidence,
158
+ diagnostics=[{"reason_codes": list(classification.reason_codes)}],
159
+ )
160
+
161
+
162
+ def _row_entries(
163
+ sheet: SheetSnapshot,
164
+ row_number: int,
165
+ min_col: int,
166
+ max_col: int,
167
+ ) -> list[tuple[str, str]]:
168
+ entries = []
169
+ for column in range(min_col, max_col + 1):
170
+ coordinate = f"{get_column_letter(column)}{row_number}"
171
+ cell = sheet.cells.get(coordinate)
172
+ if cell is None or cell.merge_anchor is not None or not cell.display_value.strip():
173
+ continue
174
+ entries.append((coordinate, cell.display_value))
175
+ return entries
176
+
177
+
178
+ def _adjacent_pairs(
179
+ entries: list[tuple[str, str]],
180
+ ) -> list[tuple[tuple[str, str], tuple[str, str]]] | None:
181
+ if len(entries) < 2 or len(entries) % 2:
182
+ return None
183
+ pairs = []
184
+ for offset in range(0, len(entries), 2):
185
+ label, value = entries[offset : offset + 2]
186
+ label_column = _column_number(label[0])
187
+ value_column = _column_number(value[0])
188
+ if value_column != label_column + 1:
189
+ return None
190
+ pairs.append((label, value))
191
+ return pairs
192
+
193
+
194
+ def _column_number(coordinate: str) -> int:
195
+ from openpyxl.utils.cell import coordinate_to_tuple
196
+
197
+ return coordinate_to_tuple(coordinate)[1]
198
+
199
+
200
+ def _display_value(sheet: SheetSnapshot, row: int, column: int) -> str:
201
+ coordinate = f"{get_column_letter(column)}{row}"
202
+ cell = sheet.cells.get(coordinate)
203
+ if cell is None or cell.merge_anchor is not None:
204
+ return ""
205
+ return cell.display_value
206
+
207
+
208
+ def _source_ref(sheet: SheetSnapshot, row: int, column: int) -> SourceRef:
209
+ return SourceRef(sheet_name=sheet.name, range=f"{get_column_letter(column)}{row}")
@@ -0,0 +1,71 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://github.com/syw2014/langparse/blob/main/langparse/workbooks/bundle-v1.schema.json",
4
+ "title": "LangParse Workbook Bundle v1",
5
+ "type": "object",
6
+ "required": ["schema_version", "facts", "structures", "relations", "diagnostics", "provenance"],
7
+ "properties": {
8
+ "schema_version": {"const": 1},
9
+ "facts": {
10
+ "type": "object",
11
+ "required": ["sheets", "references"],
12
+ "properties": {
13
+ "sheets": {
14
+ "type": "array",
15
+ "items": {
16
+ "type": "object",
17
+ "required": ["name", "visibility", "used_range", "cell_count", "cells", "objects"],
18
+ "properties": {
19
+ "name": {"type": "string", "minLength": 1},
20
+ "visibility": {"enum": ["visible", "hidden", "veryHidden"]},
21
+ "used_range": {"type": ["string", "null"]},
22
+ "cell_count": {"type": "integer", "minimum": 0},
23
+ "cells": {"type": ["array", "null"], "items": {"type": "object", "required": ["coordinate", "source_ref"], "properties": {"source_ref": {"$ref": "#/$defs/sourceRef"}}}},
24
+ "objects": {"type": "array", "items": {"type": "object"}}
25
+ }
26
+ }
27
+ },
28
+ "references": {"type": "object", "required": ["formulas", "defined_names", "tables", "external_references", "diagnostics"]}
29
+ }
30
+ },
31
+ "structures": {
32
+ "type": "object", "required": ["blocks"],
33
+ "properties": {
34
+ "blocks": {"type": "array", "items": {
35
+ "type": "object", "required": ["id", "sheet", "kind", "source_refs", "confidence", "diagnostics"],
36
+ "properties": {
37
+ "id": {"type": "string"}, "sheet": {"type": "string"},
38
+ "kind": {"enum": ["logical_table", "form", "matrix", "text", "unclassified", "chart", "image"]},
39
+ "source_refs": {"type": "array", "items": {"$ref": "#/$defs/sourceRef"}},
40
+ "confidence": {"type": "number"}, "diagnostics": {"type": "array"},
41
+ "table": {"type": "object", "required": ["table_id", "columns", "rows", "row_count", "omitted_rows", "source_refs"], "properties": {"rows": {"type": "array"}, "row_count": {"type": "integer", "minimum": 0}, "omitted_rows": {"type": "integer", "minimum": 0}}}
42
+ }
43
+ }}
44
+ }
45
+ },
46
+ "relations": {
47
+ "type": "object", "required": ["dependencies", "continuations"],
48
+ "properties": {
49
+ "dependencies": {"type": "array", "items": {
50
+ "type": "object", "required": ["dependent", "target", "status", "reference", "source_refs"],
51
+ "properties": {
52
+ "dependent": {"$ref": "#/$defs/sourceRef"},
53
+ "target": {"anyOf": [{"$ref": "#/$defs/sourceRef"}, {"type": "null"}]},
54
+ "status": {"enum": ["resolved", "external", "unresolved", "unsupported", "dynamic", "circular"]},
55
+ "reference": {"type": "string"}, "source_refs": {"type": "array", "items": {"$ref": "#/$defs/sourceRef"}}
56
+ }
57
+ }},
58
+ "continuations": {"type": "array", "items": {"type": "object", "required": ["id", "member_table_ids", "source_refs"]}}
59
+ }
60
+ },
61
+ "diagnostics": {"type": "object"},
62
+ "provenance": {
63
+ "type": "object", "required": ["source", "filename", "parser_version", "engine", "omissions"],
64
+ "properties": {"omissions": {"type": "object", "required": ["cells", "media_binary", "source_binary", "max_rows_per_block"], "properties": {
65
+ "cells": {"type": "boolean"}, "media_binary": {"const": true}, "source_binary": {"const": true},
66
+ "max_rows_per_block": {"type": ["integer", "null"], "minimum": 1}
67
+ }}}
68
+ }
69
+ },
70
+ "$defs": {"sourceRef": {"type": "string", "pattern": "^.+![A-Z]+[1-9][0-9]*(?::[A-Z]+[1-9][0-9]*)?$"}}
71
+ }
@@ -0,0 +1,341 @@
1
+ """Versioned workbook consumption interface with explicit export omissions."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from copy import deepcopy
7
+ from dataclasses import fields, is_dataclass
8
+ from typing import Any
9
+
10
+ from langparse.types import ParsedDocumentResult
11
+ from langparse.workbooks.lineage import overlaps
12
+ from langparse.workbooks.types import SourceRef, WorkbookIR
13
+
14
+ BUNDLE_SCHEMA_VERSION = 1
15
+
16
+
17
+ def _value(item: Any) -> Any:
18
+ if isinstance(item, SourceRef):
19
+ return item.key
20
+ if is_dataclass(item):
21
+ return {field.name: _value(getattr(item, field.name)) for field in fields(item)}
22
+ if isinstance(item, dict):
23
+ if set(item) == {"sheet_name", "range"}:
24
+ return f"{item['sheet_name']}!{item['range']}"
25
+ return {str(key): _value(value) for key, value in item.items()}
26
+ if isinstance(item, (list, tuple)):
27
+ return [_value(value) for value in item]
28
+ if hasattr(item, "isoformat"):
29
+ return item.isoformat()
30
+ if item is None or isinstance(item, (str, bool, int, float)):
31
+ return item
32
+ return str(item)
33
+
34
+
35
+ def _cell(sheet, cell) -> dict:
36
+ return {"source_ref": f"{sheet.name}!{cell.coordinate}", **_value(cell)}
37
+
38
+
39
+ def _edge(edge) -> dict:
40
+ return _value(edge)
41
+
42
+
43
+ class WorkbookBundle:
44
+ """Stable JSON partitions and queries over a parsed or serialized workbook."""
45
+
46
+ def __init__(self, payload: dict, parsed: ParsedDocumentResult | None = None):
47
+ if (
48
+ type(payload.get("schema_version")) is not int
49
+ or payload["schema_version"] != BUNDLE_SCHEMA_VERSION
50
+ ):
51
+ raise ValueError("Unsupported workbook bundle schema version")
52
+ for key in ("facts", "structures", "relations", "diagnostics", "provenance"):
53
+ if not isinstance(payload.get(key), dict):
54
+ raise ValueError(f"Workbook bundle {key} must be an object")
55
+ for group, key in (
56
+ ("facts", "sheets"),
57
+ ("structures", "blocks"),
58
+ ("relations", "dependencies"),
59
+ ):
60
+ if not isinstance(payload[group].get(key), list):
61
+ raise ValueError(f"Workbook bundle {group}.{key} must be an array")
62
+
63
+ def query_fields(record, required):
64
+ if not isinstance(record, dict) or not required <= record.keys():
65
+ raise ValueError("Workbook bundle record is missing a query field")
66
+
67
+ for block in payload["structures"]["blocks"]:
68
+ query_fields(block, {"sheet", "kind"})
69
+ if not all(isinstance(block[key], str) for key in ("sheet", "kind")):
70
+ raise ValueError("Workbook bundle block query fields must be strings")
71
+ for edge in payload["relations"]["dependencies"]:
72
+ query_fields(edge, {"dependent", "target"})
73
+
74
+ sheet_names = set()
75
+ for sheet in payload["facts"]["sheets"]:
76
+ if not isinstance(sheet, dict) or not isinstance(sheet.get("name"), str):
77
+ raise ValueError("Workbook bundle sheet must have a name")
78
+ if sheet["name"] in sheet_names:
79
+ raise ValueError("Duplicate workbook bundle sheet")
80
+ sheet_names.add(sheet["name"])
81
+ query_fields(sheet, {"cells"})
82
+ if sheet["cells"] is not None and not isinstance(sheet["cells"], list):
83
+ raise ValueError("Workbook bundle cells must be an array or null")
84
+ for cell in sheet["cells"] or []:
85
+ query_fields(cell, {"coordinate"})
86
+ if not isinstance(cell["coordinate"], str):
87
+ raise ValueError("Workbook bundle cell coordinate must be a string")
88
+
89
+ def source(value):
90
+ if not isinstance(value, str) or "!" not in value:
91
+ raise ValueError("Invalid workbook bundle source ref")
92
+ name, cell_range = value.rsplit("!", 1)
93
+ ref = SourceRef(name, cell_range)
94
+ if name not in sheet_names:
95
+ raise ValueError("Workbook bundle source ref uses unknown sheet")
96
+ try:
97
+ if not overlaps(ref, ref):
98
+ raise ValueError("reversed range")
99
+ except ValueError as exc:
100
+ raise ValueError("Invalid workbook bundle source range") from exc
101
+
102
+ def sources(value, *, nullable=False):
103
+ if isinstance(value, list):
104
+ for item in value:
105
+ sources(item, nullable=nullable)
106
+ elif value is not None or not nullable:
107
+ source(value)
108
+
109
+ containers = {
110
+ "facts",
111
+ "structures",
112
+ "relations",
113
+ "diagnostics",
114
+ "blocks",
115
+ "sheets",
116
+ "references",
117
+ "formulas",
118
+ "defined_names",
119
+ "tables",
120
+ "external_references",
121
+ "reference_diagnostics",
122
+ "dependencies",
123
+ "continuations",
124
+ "object",
125
+ "objects",
126
+ "series",
127
+ "cells",
128
+ "table",
129
+ "form",
130
+ "matrix",
131
+ "text",
132
+ "columns",
133
+ "rows",
134
+ "fragments",
135
+ "sections",
136
+ "fields",
137
+ "free_text",
138
+ "lines",
139
+ "row_headers",
140
+ "column_headers",
141
+ }
142
+
143
+ def refs(value):
144
+ if isinstance(value, dict):
145
+ for key, item in value.items():
146
+ if key in {"source_ref", "dependent", "target"} and item is not None:
147
+ source(item)
148
+ elif key.endswith("source_refs"):
149
+ if not isinstance(item, list):
150
+ raise ValueError("Workbook bundle source refs must be an array")
151
+ sources(item, nullable=key == "value_source_refs")
152
+ elif key in containers:
153
+ refs(item)
154
+ elif isinstance(value, list):
155
+ for item in value:
156
+ refs(item)
157
+
158
+ refs(payload)
159
+ self._payload = deepcopy(payload)
160
+ self._parsed = parsed
161
+
162
+ @classmethod
163
+ def from_result(
164
+ cls,
165
+ parsed: ParsedDocumentResult,
166
+ *,
167
+ include_cells: bool = False,
168
+ max_rows: int | None = 200,
169
+ ) -> WorkbookBundle:
170
+ if not isinstance(parsed.structure, WorkbookIR) or parsed.structure.snapshot is None:
171
+ raise TypeError("Workbook Bundle requires a rich OOXML workbook result")
172
+ if type(include_cells) is not bool:
173
+ raise ValueError("include_cells must be boolean")
174
+ if max_rows is not None and (type(max_rows) is not int or max_rows < 1):
175
+ raise ValueError("max_rows must be a positive integer or None")
176
+ workbook = parsed.structure
177
+ snapshot = workbook.snapshot
178
+ blocks = []
179
+ for sheet in workbook.sheets:
180
+ for block in sheet.blocks:
181
+ item = {
182
+ "id": block.block_id,
183
+ "sheet": sheet.name,
184
+ "kind": block.kind,
185
+ "source_refs": [ref.key for ref in block.source_refs],
186
+ "confidence": block.confidence,
187
+ "diagnostics": _value(block.diagnostics),
188
+ }
189
+ for name, content in (
190
+ ("table", block.logical_table),
191
+ ("form", block.form),
192
+ ("matrix", block.matrix),
193
+ ("text", block.text),
194
+ ):
195
+ if content is not None:
196
+ data = _value(content)
197
+ if name == "table":
198
+ data["row_count"] = len(content.rows)
199
+ data["rows"] = data["rows"][:max_rows]
200
+ data["omitted_rows"] = len(content.rows) - len(data["rows"])
201
+ elif name == "matrix":
202
+ data["row_count"] = len(content.values)
203
+ for key in ("values", "row_headers", "value_source_refs"):
204
+ data[key] = data[key][:max_rows]
205
+ data["omitted_rows"] = len(content.values) - len(data["values"])
206
+ item[name] = data
207
+ if block.kind in {"chart", "image"}:
208
+ item["object"] = _value(block.metadata["object"])
209
+ blocks.append(item)
210
+ from langparse import __version__
211
+
212
+ return cls(
213
+ {
214
+ "schema_version": BUNDLE_SCHEMA_VERSION,
215
+ "facts": {
216
+ "sheets": [
217
+ {
218
+ "name": sheet.name,
219
+ "visibility": sheet.visibility,
220
+ "used_range": sheet.used_range,
221
+ "cell_count": len(sheet.cells),
222
+ "merged_ranges": list(sheet.merged_ranges),
223
+ "cells": [_cell(sheet, cell) for cell in sheet.cells.values()]
224
+ if include_cells
225
+ else None,
226
+ "objects": _value(sheet.objects),
227
+ }
228
+ for sheet in snapshot.sheets
229
+ ],
230
+ "references": _value(snapshot.reference_facts),
231
+ },
232
+ "structures": {"blocks": blocks},
233
+ "relations": {
234
+ "dependencies": [_edge(edge) for edge in workbook.lineage.edges]
235
+ if workbook.lineage
236
+ else [],
237
+ "continuations": [
238
+ {
239
+ "id": item.continuation_id,
240
+ "member_table_ids": item.member_table_ids,
241
+ "source_refs": [ref.key for ref in item.source_refs],
242
+ }
243
+ for item in workbook.table_continuations
244
+ ],
245
+ },
246
+ "diagnostics": _value(parsed.diagnostics) if parsed.diagnostics else {},
247
+ "provenance": {
248
+ "source": parsed.source,
249
+ "filename": parsed.filename,
250
+ "parser_version": __version__,
251
+ "engine": parsed.engine,
252
+ "omissions": {
253
+ "cells": not include_cells,
254
+ "media_binary": True,
255
+ "source_binary": True,
256
+ "max_rows_per_block": max_rows,
257
+ },
258
+ },
259
+ },
260
+ parsed,
261
+ )
262
+
263
+ @classmethod
264
+ def from_json(cls, text: str) -> WorkbookBundle:
265
+ def unique(pairs):
266
+ result = {}
267
+ for key, value in pairs:
268
+ if key in result:
269
+ raise ValueError(f"Duplicate bundle field: {key}")
270
+ result[key] = value
271
+ return result
272
+
273
+ payload = json.loads(text, object_pairs_hook=unique)
274
+ if not isinstance(payload, dict):
275
+ raise ValueError("Workbook bundle root must be an object")
276
+ return cls(payload)
277
+
278
+ def to_dict(self) -> dict:
279
+ return deepcopy(self._payload)
280
+
281
+ def to_json(self) -> str:
282
+ return json.dumps(self._payload, ensure_ascii=False, sort_keys=True, indent=2)
283
+
284
+ def blocks(self, *, sheet: str | None = None, kind: str | None = None) -> list[dict]:
285
+ return deepcopy(
286
+ [
287
+ block
288
+ for block in self._payload["structures"]["blocks"]
289
+ if (sheet is None or block["sheet"] == sheet)
290
+ and (kind is None or block["kind"] == kind)
291
+ ]
292
+ )
293
+
294
+ def tables(self, *, sheet: str | None = None) -> list[dict]:
295
+ return [block["table"] for block in self.blocks(sheet=sheet) if "table" in block]
296
+
297
+ def cells(self, *, sheet: str, source_range: str | None = None) -> list[dict]:
298
+ if self._parsed is not None:
299
+ snapshot = self._parsed.structure.snapshot
300
+ selected = next((item for item in snapshot.sheets if item.name == sheet), None)
301
+ if selected is None:
302
+ raise ValueError(f"Unknown workbook sheet: {sheet}")
303
+ cells = [_cell(selected, cell) for cell in selected.cells.values()]
304
+ else:
305
+ selected = next(
306
+ (item for item in self._payload["facts"]["sheets"] if item["name"] == sheet), None
307
+ )
308
+ if selected is None:
309
+ raise ValueError(f"Unknown workbook sheet: {sheet}")
310
+ cells = selected["cells"]
311
+ if cells is None:
312
+ raise ValueError(
313
+ "Cells were omitted; export with include_cells=True to query a loaded bundle"
314
+ )
315
+ if source_range is not None:
316
+ region = SourceRef(sheet, source_range)
317
+ cells = [
318
+ cell for cell in cells if overlaps(region, SourceRef(sheet, cell["coordinate"]))
319
+ ]
320
+ return deepcopy(cells)
321
+
322
+ def relations(
323
+ self, *, source_ref: str | None = None, direction: str = "dependencies"
324
+ ) -> list[dict]:
325
+ if direction not in {"dependencies", "dependents"}:
326
+ raise ValueError("direction must be dependencies or dependents")
327
+ edges = self._payload["relations"]["dependencies"]
328
+ if source_ref is not None:
329
+ name, cell_range = source_ref.rsplit("!", 1)
330
+ selected = SourceRef(name, cell_range)
331
+ key = "dependent" if direction == "dependencies" else "target"
332
+ edges = [
333
+ edge
334
+ for edge in edges
335
+ if edge[key] and overlaps(selected, SourceRef(*edge[key].rsplit("!", 1)))
336
+ ]
337
+ return deepcopy(edges)
338
+
339
+ @property
340
+ def diagnostics(self) -> dict:
341
+ return deepcopy(self._payload["diagnostics"])