langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from openpyxl.utils import get_column_letter, range_boundaries
|
|
4
|
+
|
|
5
|
+
from langparse.workbooks.classification import BlockClassification
|
|
6
|
+
from langparse.workbooks.types import (
|
|
7
|
+
CandidateRegion,
|
|
8
|
+
FormBlock,
|
|
9
|
+
FormField,
|
|
10
|
+
MatrixBlock,
|
|
11
|
+
MatrixHeader,
|
|
12
|
+
SheetSnapshot,
|
|
13
|
+
SourceRef,
|
|
14
|
+
TextBlock,
|
|
15
|
+
TextLine,
|
|
16
|
+
stable_id,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def interpret_form_block(
|
|
21
|
+
sheet: SheetSnapshot,
|
|
22
|
+
candidate: CandidateRegion,
|
|
23
|
+
classification: BlockClassification,
|
|
24
|
+
) -> FormBlock:
|
|
25
|
+
"""Interpret adjacent label/value pairs without inventing missing fields."""
|
|
26
|
+
|
|
27
|
+
min_col, min_row, max_col, max_row = range_boundaries(candidate.source_ref.range)
|
|
28
|
+
title = ""
|
|
29
|
+
fields: list[FormField] = []
|
|
30
|
+
free_text: list[TextLine] = []
|
|
31
|
+
for row_number in range(min_row, max_row + 1):
|
|
32
|
+
entries = _row_entries(sheet, row_number, min_col, max_col)
|
|
33
|
+
if row_number == min_row and len(entries) == 1:
|
|
34
|
+
title = entries[0][1]
|
|
35
|
+
continue
|
|
36
|
+
pairs = _adjacent_pairs(entries)
|
|
37
|
+
if pairs is not None:
|
|
38
|
+
for (label_coordinate, label), (value_coordinate, value) in pairs:
|
|
39
|
+
label_ref = SourceRef(sheet_name=sheet.name, range=label_coordinate)
|
|
40
|
+
value_ref = SourceRef(sheet_name=sheet.name, range=value_coordinate)
|
|
41
|
+
fields.append(
|
|
42
|
+
FormField(
|
|
43
|
+
field_id=stable_id(
|
|
44
|
+
"field",
|
|
45
|
+
candidate.source_ref.key,
|
|
46
|
+
label_ref.key,
|
|
47
|
+
value_ref.key,
|
|
48
|
+
),
|
|
49
|
+
label=label,
|
|
50
|
+
value=value,
|
|
51
|
+
label_source_refs=[label_ref],
|
|
52
|
+
value_source_refs=[value_ref],
|
|
53
|
+
confidence=classification.confidence,
|
|
54
|
+
)
|
|
55
|
+
)
|
|
56
|
+
continue
|
|
57
|
+
if entries:
|
|
58
|
+
free_text.append(
|
|
59
|
+
TextLine(
|
|
60
|
+
text=" ".join(value for _, value in entries),
|
|
61
|
+
source_refs=[
|
|
62
|
+
SourceRef(sheet_name=sheet.name, range=coordinate)
|
|
63
|
+
for coordinate, _ in entries
|
|
64
|
+
],
|
|
65
|
+
)
|
|
66
|
+
)
|
|
67
|
+
return FormBlock(
|
|
68
|
+
form_id=stable_id("form", candidate.source_ref.key),
|
|
69
|
+
title=title,
|
|
70
|
+
fields=fields,
|
|
71
|
+
free_text=free_text,
|
|
72
|
+
source_refs=[candidate.source_ref],
|
|
73
|
+
confidence=classification.confidence,
|
|
74
|
+
diagnostics=[{"reason_codes": list(classification.reason_codes)}],
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def interpret_matrix_block(
|
|
79
|
+
sheet: SheetSnapshot,
|
|
80
|
+
candidate: CandidateRegion,
|
|
81
|
+
classification: BlockClassification,
|
|
82
|
+
) -> MatrixBlock:
|
|
83
|
+
"""Preserve a two-axis matrix and its physical value grid."""
|
|
84
|
+
|
|
85
|
+
min_col, min_row, max_col, max_row = range_boundaries(candidate.source_ref.range)
|
|
86
|
+
if max_col - min_col < 2 or max_row - min_row < 2:
|
|
87
|
+
raise ValueError("matrix requires at least two row and column dimensions")
|
|
88
|
+
|
|
89
|
+
title = _display_value(sheet, min_row, min_col)
|
|
90
|
+
column_headers = [
|
|
91
|
+
MatrixHeader(
|
|
92
|
+
value=_display_value(sheet, min_row, column),
|
|
93
|
+
source_refs=[_source_ref(sheet, min_row, column)],
|
|
94
|
+
)
|
|
95
|
+
for column in range(min_col + 1, max_col + 1)
|
|
96
|
+
]
|
|
97
|
+
row_headers = [
|
|
98
|
+
MatrixHeader(
|
|
99
|
+
value=_display_value(sheet, row, min_col),
|
|
100
|
+
source_refs=[_source_ref(sheet, row, min_col)],
|
|
101
|
+
)
|
|
102
|
+
for row in range(min_row + 1, max_row + 1)
|
|
103
|
+
]
|
|
104
|
+
if any(not header.value for header in [*column_headers, *row_headers]):
|
|
105
|
+
raise ValueError("matrix axes must be complete")
|
|
106
|
+
|
|
107
|
+
values = []
|
|
108
|
+
value_source_refs = []
|
|
109
|
+
for row in range(min_row + 1, max_row + 1):
|
|
110
|
+
value_row = []
|
|
111
|
+
ref_row = []
|
|
112
|
+
for column in range(min_col + 1, max_col + 1):
|
|
113
|
+
value_row.append(_display_value(sheet, row, column))
|
|
114
|
+
coordinate = f"{get_column_letter(column)}{row}"
|
|
115
|
+
ref_row.append(_source_ref(sheet, row, column) if coordinate in sheet.cells else None)
|
|
116
|
+
values.append(value_row)
|
|
117
|
+
value_source_refs.append(ref_row)
|
|
118
|
+
|
|
119
|
+
return MatrixBlock(
|
|
120
|
+
matrix_id=stable_id("matrix", candidate.source_ref.key),
|
|
121
|
+
title=title,
|
|
122
|
+
row_headers=row_headers,
|
|
123
|
+
column_headers=column_headers,
|
|
124
|
+
values=values,
|
|
125
|
+
source_refs=[candidate.source_ref],
|
|
126
|
+
value_source_refs=value_source_refs,
|
|
127
|
+
confidence=classification.confidence,
|
|
128
|
+
diagnostics=[{"reason_codes": list(classification.reason_codes)}],
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def interpret_text_block(
|
|
133
|
+
sheet: SheetSnapshot,
|
|
134
|
+
candidate: CandidateRegion,
|
|
135
|
+
classification: BlockClassification,
|
|
136
|
+
) -> TextBlock:
|
|
137
|
+
"""Render source-ordered display text without merged-cell duplication."""
|
|
138
|
+
|
|
139
|
+
min_col, min_row, max_col, max_row = range_boundaries(candidate.source_ref.range)
|
|
140
|
+
lines = []
|
|
141
|
+
for row_number in range(min_row, max_row + 1):
|
|
142
|
+
entries = _row_entries(sheet, row_number, min_col, max_col)
|
|
143
|
+
if entries:
|
|
144
|
+
lines.append(
|
|
145
|
+
TextLine(
|
|
146
|
+
text=" ".join(value for _, value in entries),
|
|
147
|
+
source_refs=[
|
|
148
|
+
SourceRef(sheet_name=sheet.name, range=coordinate)
|
|
149
|
+
for coordinate, _ in entries
|
|
150
|
+
],
|
|
151
|
+
)
|
|
152
|
+
)
|
|
153
|
+
return TextBlock(
|
|
154
|
+
text_id=stable_id("text", candidate.source_ref.key),
|
|
155
|
+
lines=lines,
|
|
156
|
+
source_refs=[candidate.source_ref],
|
|
157
|
+
confidence=classification.confidence,
|
|
158
|
+
diagnostics=[{"reason_codes": list(classification.reason_codes)}],
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _row_entries(
|
|
163
|
+
sheet: SheetSnapshot,
|
|
164
|
+
row_number: int,
|
|
165
|
+
min_col: int,
|
|
166
|
+
max_col: int,
|
|
167
|
+
) -> list[tuple[str, str]]:
|
|
168
|
+
entries = []
|
|
169
|
+
for column in range(min_col, max_col + 1):
|
|
170
|
+
coordinate = f"{get_column_letter(column)}{row_number}"
|
|
171
|
+
cell = sheet.cells.get(coordinate)
|
|
172
|
+
if cell is None or cell.merge_anchor is not None or not cell.display_value.strip():
|
|
173
|
+
continue
|
|
174
|
+
entries.append((coordinate, cell.display_value))
|
|
175
|
+
return entries
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _adjacent_pairs(
|
|
179
|
+
entries: list[tuple[str, str]],
|
|
180
|
+
) -> list[tuple[tuple[str, str], tuple[str, str]]] | None:
|
|
181
|
+
if len(entries) < 2 or len(entries) % 2:
|
|
182
|
+
return None
|
|
183
|
+
pairs = []
|
|
184
|
+
for offset in range(0, len(entries), 2):
|
|
185
|
+
label, value = entries[offset : offset + 2]
|
|
186
|
+
label_column = _column_number(label[0])
|
|
187
|
+
value_column = _column_number(value[0])
|
|
188
|
+
if value_column != label_column + 1:
|
|
189
|
+
return None
|
|
190
|
+
pairs.append((label, value))
|
|
191
|
+
return pairs
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _column_number(coordinate: str) -> int:
|
|
195
|
+
from openpyxl.utils.cell import coordinate_to_tuple
|
|
196
|
+
|
|
197
|
+
return coordinate_to_tuple(coordinate)[1]
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _display_value(sheet: SheetSnapshot, row: int, column: int) -> str:
|
|
201
|
+
coordinate = f"{get_column_letter(column)}{row}"
|
|
202
|
+
cell = sheet.cells.get(coordinate)
|
|
203
|
+
if cell is None or cell.merge_anchor is not None:
|
|
204
|
+
return ""
|
|
205
|
+
return cell.display_value
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _source_ref(sheet: SheetSnapshot, row: int, column: int) -> SourceRef:
|
|
209
|
+
return SourceRef(sheet_name=sheet.name, range=f"{get_column_letter(column)}{row}")
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://github.com/syw2014/langparse/blob/main/langparse/workbooks/bundle-v1.schema.json",
|
|
4
|
+
"title": "LangParse Workbook Bundle v1",
|
|
5
|
+
"type": "object",
|
|
6
|
+
"required": ["schema_version", "facts", "structures", "relations", "diagnostics", "provenance"],
|
|
7
|
+
"properties": {
|
|
8
|
+
"schema_version": {"const": 1},
|
|
9
|
+
"facts": {
|
|
10
|
+
"type": "object",
|
|
11
|
+
"required": ["sheets", "references"],
|
|
12
|
+
"properties": {
|
|
13
|
+
"sheets": {
|
|
14
|
+
"type": "array",
|
|
15
|
+
"items": {
|
|
16
|
+
"type": "object",
|
|
17
|
+
"required": ["name", "visibility", "used_range", "cell_count", "cells", "objects"],
|
|
18
|
+
"properties": {
|
|
19
|
+
"name": {"type": "string", "minLength": 1},
|
|
20
|
+
"visibility": {"enum": ["visible", "hidden", "veryHidden"]},
|
|
21
|
+
"used_range": {"type": ["string", "null"]},
|
|
22
|
+
"cell_count": {"type": "integer", "minimum": 0},
|
|
23
|
+
"cells": {"type": ["array", "null"], "items": {"type": "object", "required": ["coordinate", "source_ref"], "properties": {"source_ref": {"$ref": "#/$defs/sourceRef"}}}},
|
|
24
|
+
"objects": {"type": "array", "items": {"type": "object"}}
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
},
|
|
28
|
+
"references": {"type": "object", "required": ["formulas", "defined_names", "tables", "external_references", "diagnostics"]}
|
|
29
|
+
}
|
|
30
|
+
},
|
|
31
|
+
"structures": {
|
|
32
|
+
"type": "object", "required": ["blocks"],
|
|
33
|
+
"properties": {
|
|
34
|
+
"blocks": {"type": "array", "items": {
|
|
35
|
+
"type": "object", "required": ["id", "sheet", "kind", "source_refs", "confidence", "diagnostics"],
|
|
36
|
+
"properties": {
|
|
37
|
+
"id": {"type": "string"}, "sheet": {"type": "string"},
|
|
38
|
+
"kind": {"enum": ["logical_table", "form", "matrix", "text", "unclassified", "chart", "image"]},
|
|
39
|
+
"source_refs": {"type": "array", "items": {"$ref": "#/$defs/sourceRef"}},
|
|
40
|
+
"confidence": {"type": "number"}, "diagnostics": {"type": "array"},
|
|
41
|
+
"table": {"type": "object", "required": ["table_id", "columns", "rows", "row_count", "omitted_rows", "source_refs"], "properties": {"rows": {"type": "array"}, "row_count": {"type": "integer", "minimum": 0}, "omitted_rows": {"type": "integer", "minimum": 0}}}
|
|
42
|
+
}
|
|
43
|
+
}}
|
|
44
|
+
}
|
|
45
|
+
},
|
|
46
|
+
"relations": {
|
|
47
|
+
"type": "object", "required": ["dependencies", "continuations"],
|
|
48
|
+
"properties": {
|
|
49
|
+
"dependencies": {"type": "array", "items": {
|
|
50
|
+
"type": "object", "required": ["dependent", "target", "status", "reference", "source_refs"],
|
|
51
|
+
"properties": {
|
|
52
|
+
"dependent": {"$ref": "#/$defs/sourceRef"},
|
|
53
|
+
"target": {"anyOf": [{"$ref": "#/$defs/sourceRef"}, {"type": "null"}]},
|
|
54
|
+
"status": {"enum": ["resolved", "external", "unresolved", "unsupported", "dynamic", "circular"]},
|
|
55
|
+
"reference": {"type": "string"}, "source_refs": {"type": "array", "items": {"$ref": "#/$defs/sourceRef"}}
|
|
56
|
+
}
|
|
57
|
+
}},
|
|
58
|
+
"continuations": {"type": "array", "items": {"type": "object", "required": ["id", "member_table_ids", "source_refs"]}}
|
|
59
|
+
}
|
|
60
|
+
},
|
|
61
|
+
"diagnostics": {"type": "object"},
|
|
62
|
+
"provenance": {
|
|
63
|
+
"type": "object", "required": ["source", "filename", "parser_version", "engine", "omissions"],
|
|
64
|
+
"properties": {"omissions": {"type": "object", "required": ["cells", "media_binary", "source_binary", "max_rows_per_block"], "properties": {
|
|
65
|
+
"cells": {"type": "boolean"}, "media_binary": {"const": true}, "source_binary": {"const": true},
|
|
66
|
+
"max_rows_per_block": {"type": ["integer", "null"], "minimum": 1}
|
|
67
|
+
}}}
|
|
68
|
+
}
|
|
69
|
+
},
|
|
70
|
+
"$defs": {"sourceRef": {"type": "string", "pattern": "^.+![A-Z]+[1-9][0-9]*(?::[A-Z]+[1-9][0-9]*)?$"}}
|
|
71
|
+
}
|
|
@@ -0,0 +1,341 @@
|
|
|
1
|
+
"""Versioned workbook consumption interface with explicit export omissions."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from copy import deepcopy
|
|
7
|
+
from dataclasses import fields, is_dataclass
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
from langparse.types import ParsedDocumentResult
|
|
11
|
+
from langparse.workbooks.lineage import overlaps
|
|
12
|
+
from langparse.workbooks.types import SourceRef, WorkbookIR
|
|
13
|
+
|
|
14
|
+
BUNDLE_SCHEMA_VERSION = 1
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _value(item: Any) -> Any:
|
|
18
|
+
if isinstance(item, SourceRef):
|
|
19
|
+
return item.key
|
|
20
|
+
if is_dataclass(item):
|
|
21
|
+
return {field.name: _value(getattr(item, field.name)) for field in fields(item)}
|
|
22
|
+
if isinstance(item, dict):
|
|
23
|
+
if set(item) == {"sheet_name", "range"}:
|
|
24
|
+
return f"{item['sheet_name']}!{item['range']}"
|
|
25
|
+
return {str(key): _value(value) for key, value in item.items()}
|
|
26
|
+
if isinstance(item, (list, tuple)):
|
|
27
|
+
return [_value(value) for value in item]
|
|
28
|
+
if hasattr(item, "isoformat"):
|
|
29
|
+
return item.isoformat()
|
|
30
|
+
if item is None or isinstance(item, (str, bool, int, float)):
|
|
31
|
+
return item
|
|
32
|
+
return str(item)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _cell(sheet, cell) -> dict:
|
|
36
|
+
return {"source_ref": f"{sheet.name}!{cell.coordinate}", **_value(cell)}
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _edge(edge) -> dict:
|
|
40
|
+
return _value(edge)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class WorkbookBundle:
|
|
44
|
+
"""Stable JSON partitions and queries over a parsed or serialized workbook."""
|
|
45
|
+
|
|
46
|
+
def __init__(self, payload: dict, parsed: ParsedDocumentResult | None = None):
|
|
47
|
+
if (
|
|
48
|
+
type(payload.get("schema_version")) is not int
|
|
49
|
+
or payload["schema_version"] != BUNDLE_SCHEMA_VERSION
|
|
50
|
+
):
|
|
51
|
+
raise ValueError("Unsupported workbook bundle schema version")
|
|
52
|
+
for key in ("facts", "structures", "relations", "diagnostics", "provenance"):
|
|
53
|
+
if not isinstance(payload.get(key), dict):
|
|
54
|
+
raise ValueError(f"Workbook bundle {key} must be an object")
|
|
55
|
+
for group, key in (
|
|
56
|
+
("facts", "sheets"),
|
|
57
|
+
("structures", "blocks"),
|
|
58
|
+
("relations", "dependencies"),
|
|
59
|
+
):
|
|
60
|
+
if not isinstance(payload[group].get(key), list):
|
|
61
|
+
raise ValueError(f"Workbook bundle {group}.{key} must be an array")
|
|
62
|
+
|
|
63
|
+
def query_fields(record, required):
|
|
64
|
+
if not isinstance(record, dict) or not required <= record.keys():
|
|
65
|
+
raise ValueError("Workbook bundle record is missing a query field")
|
|
66
|
+
|
|
67
|
+
for block in payload["structures"]["blocks"]:
|
|
68
|
+
query_fields(block, {"sheet", "kind"})
|
|
69
|
+
if not all(isinstance(block[key], str) for key in ("sheet", "kind")):
|
|
70
|
+
raise ValueError("Workbook bundle block query fields must be strings")
|
|
71
|
+
for edge in payload["relations"]["dependencies"]:
|
|
72
|
+
query_fields(edge, {"dependent", "target"})
|
|
73
|
+
|
|
74
|
+
sheet_names = set()
|
|
75
|
+
for sheet in payload["facts"]["sheets"]:
|
|
76
|
+
if not isinstance(sheet, dict) or not isinstance(sheet.get("name"), str):
|
|
77
|
+
raise ValueError("Workbook bundle sheet must have a name")
|
|
78
|
+
if sheet["name"] in sheet_names:
|
|
79
|
+
raise ValueError("Duplicate workbook bundle sheet")
|
|
80
|
+
sheet_names.add(sheet["name"])
|
|
81
|
+
query_fields(sheet, {"cells"})
|
|
82
|
+
if sheet["cells"] is not None and not isinstance(sheet["cells"], list):
|
|
83
|
+
raise ValueError("Workbook bundle cells must be an array or null")
|
|
84
|
+
for cell in sheet["cells"] or []:
|
|
85
|
+
query_fields(cell, {"coordinate"})
|
|
86
|
+
if not isinstance(cell["coordinate"], str):
|
|
87
|
+
raise ValueError("Workbook bundle cell coordinate must be a string")
|
|
88
|
+
|
|
89
|
+
def source(value):
|
|
90
|
+
if not isinstance(value, str) or "!" not in value:
|
|
91
|
+
raise ValueError("Invalid workbook bundle source ref")
|
|
92
|
+
name, cell_range = value.rsplit("!", 1)
|
|
93
|
+
ref = SourceRef(name, cell_range)
|
|
94
|
+
if name not in sheet_names:
|
|
95
|
+
raise ValueError("Workbook bundle source ref uses unknown sheet")
|
|
96
|
+
try:
|
|
97
|
+
if not overlaps(ref, ref):
|
|
98
|
+
raise ValueError("reversed range")
|
|
99
|
+
except ValueError as exc:
|
|
100
|
+
raise ValueError("Invalid workbook bundle source range") from exc
|
|
101
|
+
|
|
102
|
+
def sources(value, *, nullable=False):
|
|
103
|
+
if isinstance(value, list):
|
|
104
|
+
for item in value:
|
|
105
|
+
sources(item, nullable=nullable)
|
|
106
|
+
elif value is not None or not nullable:
|
|
107
|
+
source(value)
|
|
108
|
+
|
|
109
|
+
containers = {
|
|
110
|
+
"facts",
|
|
111
|
+
"structures",
|
|
112
|
+
"relations",
|
|
113
|
+
"diagnostics",
|
|
114
|
+
"blocks",
|
|
115
|
+
"sheets",
|
|
116
|
+
"references",
|
|
117
|
+
"formulas",
|
|
118
|
+
"defined_names",
|
|
119
|
+
"tables",
|
|
120
|
+
"external_references",
|
|
121
|
+
"reference_diagnostics",
|
|
122
|
+
"dependencies",
|
|
123
|
+
"continuations",
|
|
124
|
+
"object",
|
|
125
|
+
"objects",
|
|
126
|
+
"series",
|
|
127
|
+
"cells",
|
|
128
|
+
"table",
|
|
129
|
+
"form",
|
|
130
|
+
"matrix",
|
|
131
|
+
"text",
|
|
132
|
+
"columns",
|
|
133
|
+
"rows",
|
|
134
|
+
"fragments",
|
|
135
|
+
"sections",
|
|
136
|
+
"fields",
|
|
137
|
+
"free_text",
|
|
138
|
+
"lines",
|
|
139
|
+
"row_headers",
|
|
140
|
+
"column_headers",
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
def refs(value):
|
|
144
|
+
if isinstance(value, dict):
|
|
145
|
+
for key, item in value.items():
|
|
146
|
+
if key in {"source_ref", "dependent", "target"} and item is not None:
|
|
147
|
+
source(item)
|
|
148
|
+
elif key.endswith("source_refs"):
|
|
149
|
+
if not isinstance(item, list):
|
|
150
|
+
raise ValueError("Workbook bundle source refs must be an array")
|
|
151
|
+
sources(item, nullable=key == "value_source_refs")
|
|
152
|
+
elif key in containers:
|
|
153
|
+
refs(item)
|
|
154
|
+
elif isinstance(value, list):
|
|
155
|
+
for item in value:
|
|
156
|
+
refs(item)
|
|
157
|
+
|
|
158
|
+
refs(payload)
|
|
159
|
+
self._payload = deepcopy(payload)
|
|
160
|
+
self._parsed = parsed
|
|
161
|
+
|
|
162
|
+
@classmethod
|
|
163
|
+
def from_result(
|
|
164
|
+
cls,
|
|
165
|
+
parsed: ParsedDocumentResult,
|
|
166
|
+
*,
|
|
167
|
+
include_cells: bool = False,
|
|
168
|
+
max_rows: int | None = 200,
|
|
169
|
+
) -> WorkbookBundle:
|
|
170
|
+
if not isinstance(parsed.structure, WorkbookIR) or parsed.structure.snapshot is None:
|
|
171
|
+
raise TypeError("Workbook Bundle requires a rich OOXML workbook result")
|
|
172
|
+
if type(include_cells) is not bool:
|
|
173
|
+
raise ValueError("include_cells must be boolean")
|
|
174
|
+
if max_rows is not None and (type(max_rows) is not int or max_rows < 1):
|
|
175
|
+
raise ValueError("max_rows must be a positive integer or None")
|
|
176
|
+
workbook = parsed.structure
|
|
177
|
+
snapshot = workbook.snapshot
|
|
178
|
+
blocks = []
|
|
179
|
+
for sheet in workbook.sheets:
|
|
180
|
+
for block in sheet.blocks:
|
|
181
|
+
item = {
|
|
182
|
+
"id": block.block_id,
|
|
183
|
+
"sheet": sheet.name,
|
|
184
|
+
"kind": block.kind,
|
|
185
|
+
"source_refs": [ref.key for ref in block.source_refs],
|
|
186
|
+
"confidence": block.confidence,
|
|
187
|
+
"diagnostics": _value(block.diagnostics),
|
|
188
|
+
}
|
|
189
|
+
for name, content in (
|
|
190
|
+
("table", block.logical_table),
|
|
191
|
+
("form", block.form),
|
|
192
|
+
("matrix", block.matrix),
|
|
193
|
+
("text", block.text),
|
|
194
|
+
):
|
|
195
|
+
if content is not None:
|
|
196
|
+
data = _value(content)
|
|
197
|
+
if name == "table":
|
|
198
|
+
data["row_count"] = len(content.rows)
|
|
199
|
+
data["rows"] = data["rows"][:max_rows]
|
|
200
|
+
data["omitted_rows"] = len(content.rows) - len(data["rows"])
|
|
201
|
+
elif name == "matrix":
|
|
202
|
+
data["row_count"] = len(content.values)
|
|
203
|
+
for key in ("values", "row_headers", "value_source_refs"):
|
|
204
|
+
data[key] = data[key][:max_rows]
|
|
205
|
+
data["omitted_rows"] = len(content.values) - len(data["values"])
|
|
206
|
+
item[name] = data
|
|
207
|
+
if block.kind in {"chart", "image"}:
|
|
208
|
+
item["object"] = _value(block.metadata["object"])
|
|
209
|
+
blocks.append(item)
|
|
210
|
+
from langparse import __version__
|
|
211
|
+
|
|
212
|
+
return cls(
|
|
213
|
+
{
|
|
214
|
+
"schema_version": BUNDLE_SCHEMA_VERSION,
|
|
215
|
+
"facts": {
|
|
216
|
+
"sheets": [
|
|
217
|
+
{
|
|
218
|
+
"name": sheet.name,
|
|
219
|
+
"visibility": sheet.visibility,
|
|
220
|
+
"used_range": sheet.used_range,
|
|
221
|
+
"cell_count": len(sheet.cells),
|
|
222
|
+
"merged_ranges": list(sheet.merged_ranges),
|
|
223
|
+
"cells": [_cell(sheet, cell) for cell in sheet.cells.values()]
|
|
224
|
+
if include_cells
|
|
225
|
+
else None,
|
|
226
|
+
"objects": _value(sheet.objects),
|
|
227
|
+
}
|
|
228
|
+
for sheet in snapshot.sheets
|
|
229
|
+
],
|
|
230
|
+
"references": _value(snapshot.reference_facts),
|
|
231
|
+
},
|
|
232
|
+
"structures": {"blocks": blocks},
|
|
233
|
+
"relations": {
|
|
234
|
+
"dependencies": [_edge(edge) for edge in workbook.lineage.edges]
|
|
235
|
+
if workbook.lineage
|
|
236
|
+
else [],
|
|
237
|
+
"continuations": [
|
|
238
|
+
{
|
|
239
|
+
"id": item.continuation_id,
|
|
240
|
+
"member_table_ids": item.member_table_ids,
|
|
241
|
+
"source_refs": [ref.key for ref in item.source_refs],
|
|
242
|
+
}
|
|
243
|
+
for item in workbook.table_continuations
|
|
244
|
+
],
|
|
245
|
+
},
|
|
246
|
+
"diagnostics": _value(parsed.diagnostics) if parsed.diagnostics else {},
|
|
247
|
+
"provenance": {
|
|
248
|
+
"source": parsed.source,
|
|
249
|
+
"filename": parsed.filename,
|
|
250
|
+
"parser_version": __version__,
|
|
251
|
+
"engine": parsed.engine,
|
|
252
|
+
"omissions": {
|
|
253
|
+
"cells": not include_cells,
|
|
254
|
+
"media_binary": True,
|
|
255
|
+
"source_binary": True,
|
|
256
|
+
"max_rows_per_block": max_rows,
|
|
257
|
+
},
|
|
258
|
+
},
|
|
259
|
+
},
|
|
260
|
+
parsed,
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
@classmethod
|
|
264
|
+
def from_json(cls, text: str) -> WorkbookBundle:
|
|
265
|
+
def unique(pairs):
|
|
266
|
+
result = {}
|
|
267
|
+
for key, value in pairs:
|
|
268
|
+
if key in result:
|
|
269
|
+
raise ValueError(f"Duplicate bundle field: {key}")
|
|
270
|
+
result[key] = value
|
|
271
|
+
return result
|
|
272
|
+
|
|
273
|
+
payload = json.loads(text, object_pairs_hook=unique)
|
|
274
|
+
if not isinstance(payload, dict):
|
|
275
|
+
raise ValueError("Workbook bundle root must be an object")
|
|
276
|
+
return cls(payload)
|
|
277
|
+
|
|
278
|
+
def to_dict(self) -> dict:
|
|
279
|
+
return deepcopy(self._payload)
|
|
280
|
+
|
|
281
|
+
def to_json(self) -> str:
|
|
282
|
+
return json.dumps(self._payload, ensure_ascii=False, sort_keys=True, indent=2)
|
|
283
|
+
|
|
284
|
+
def blocks(self, *, sheet: str | None = None, kind: str | None = None) -> list[dict]:
|
|
285
|
+
return deepcopy(
|
|
286
|
+
[
|
|
287
|
+
block
|
|
288
|
+
for block in self._payload["structures"]["blocks"]
|
|
289
|
+
if (sheet is None or block["sheet"] == sheet)
|
|
290
|
+
and (kind is None or block["kind"] == kind)
|
|
291
|
+
]
|
|
292
|
+
)
|
|
293
|
+
|
|
294
|
+
def tables(self, *, sheet: str | None = None) -> list[dict]:
|
|
295
|
+
return [block["table"] for block in self.blocks(sheet=sheet) if "table" in block]
|
|
296
|
+
|
|
297
|
+
def cells(self, *, sheet: str, source_range: str | None = None) -> list[dict]:
|
|
298
|
+
if self._parsed is not None:
|
|
299
|
+
snapshot = self._parsed.structure.snapshot
|
|
300
|
+
selected = next((item for item in snapshot.sheets if item.name == sheet), None)
|
|
301
|
+
if selected is None:
|
|
302
|
+
raise ValueError(f"Unknown workbook sheet: {sheet}")
|
|
303
|
+
cells = [_cell(selected, cell) for cell in selected.cells.values()]
|
|
304
|
+
else:
|
|
305
|
+
selected = next(
|
|
306
|
+
(item for item in self._payload["facts"]["sheets"] if item["name"] == sheet), None
|
|
307
|
+
)
|
|
308
|
+
if selected is None:
|
|
309
|
+
raise ValueError(f"Unknown workbook sheet: {sheet}")
|
|
310
|
+
cells = selected["cells"]
|
|
311
|
+
if cells is None:
|
|
312
|
+
raise ValueError(
|
|
313
|
+
"Cells were omitted; export with include_cells=True to query a loaded bundle"
|
|
314
|
+
)
|
|
315
|
+
if source_range is not None:
|
|
316
|
+
region = SourceRef(sheet, source_range)
|
|
317
|
+
cells = [
|
|
318
|
+
cell for cell in cells if overlaps(region, SourceRef(sheet, cell["coordinate"]))
|
|
319
|
+
]
|
|
320
|
+
return deepcopy(cells)
|
|
321
|
+
|
|
322
|
+
def relations(
|
|
323
|
+
self, *, source_ref: str | None = None, direction: str = "dependencies"
|
|
324
|
+
) -> list[dict]:
|
|
325
|
+
if direction not in {"dependencies", "dependents"}:
|
|
326
|
+
raise ValueError("direction must be dependencies or dependents")
|
|
327
|
+
edges = self._payload["relations"]["dependencies"]
|
|
328
|
+
if source_ref is not None:
|
|
329
|
+
name, cell_range = source_ref.rsplit("!", 1)
|
|
330
|
+
selected = SourceRef(name, cell_range)
|
|
331
|
+
key = "dependent" if direction == "dependencies" else "target"
|
|
332
|
+
edges = [
|
|
333
|
+
edge
|
|
334
|
+
for edge in edges
|
|
335
|
+
if edge[key] and overlaps(selected, SourceRef(*edge[key].rsplit("!", 1)))
|
|
336
|
+
]
|
|
337
|
+
return deepcopy(edges)
|
|
338
|
+
|
|
339
|
+
@property
|
|
340
|
+
def diagnostics(self) -> dict:
|
|
341
|
+
return deepcopy(self._payload["diagnostics"])
|