langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
"""Conservative Excel reference parsing; never evaluates a formula or opens a link."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
|
|
7
|
+
from openpyxl.formula.tokenizer import Tokenizer, TokenizerError
|
|
8
|
+
from openpyxl.utils import get_column_letter, range_boundaries
|
|
9
|
+
|
|
10
|
+
from langparse.workbooks.reference_types import (
|
|
11
|
+
ExternalReferenceFact,
|
|
12
|
+
FormulaFact,
|
|
13
|
+
FormulaReference,
|
|
14
|
+
ReferenceDiagnostic,
|
|
15
|
+
)
|
|
16
|
+
from langparse.workbooks.types import SourceRef, WorkbookSnapshot
|
|
17
|
+
|
|
18
|
+
_A1 = re.compile(r"\$?[A-Za-z]{1,3}\$?[1-9][0-9]*(?::\$?[A-Za-z]{1,3}\$?[1-9][0-9]*)?$")
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def extract_reference_facts(snapshot: WorkbookSnapshot) -> None:
|
|
22
|
+
facts = snapshot.reference_facts
|
|
23
|
+
facts.formulas.clear()
|
|
24
|
+
facts.external_references.clear()
|
|
25
|
+
facts.diagnostics.clear()
|
|
26
|
+
for sheet in snapshot.sheets:
|
|
27
|
+
for cell in sheet.cells.values():
|
|
28
|
+
if cell.formula is None:
|
|
29
|
+
continue
|
|
30
|
+
source = SourceRef(sheet.name, cell.coordinate)
|
|
31
|
+
if not isinstance(cell.formula, str):
|
|
32
|
+
facts.diagnostics.append(
|
|
33
|
+
ReferenceDiagnostic(
|
|
34
|
+
"unsupported_formula", "Non-text formula representation", [source]
|
|
35
|
+
)
|
|
36
|
+
)
|
|
37
|
+
continue
|
|
38
|
+
fact = FormulaFact(source, cell.formula)
|
|
39
|
+
facts.formulas.append(fact)
|
|
40
|
+
try:
|
|
41
|
+
tokens = Tokenizer(cell.formula).items
|
|
42
|
+
except (TokenizerError, IndexError, ValueError):
|
|
43
|
+
facts.diagnostics.append(
|
|
44
|
+
ReferenceDiagnostic("unsupported_formula", cell.formula, [source])
|
|
45
|
+
)
|
|
46
|
+
continue
|
|
47
|
+
for token in tokens:
|
|
48
|
+
if token.type == "FUNC" and token.subtype == "OPEN":
|
|
49
|
+
if token.value[:-1].upper().removeprefix("_XLFN.") in {"INDIRECT", "OFFSET"}:
|
|
50
|
+
facts.diagnostics.append(
|
|
51
|
+
ReferenceDiagnostic("dynamic_reference", token.value[:-1], [source])
|
|
52
|
+
)
|
|
53
|
+
if token.type == "OPERAND" and token.subtype == "ERROR":
|
|
54
|
+
facts.diagnostics.append(
|
|
55
|
+
ReferenceDiagnostic("invalid_reference", token.value, [source])
|
|
56
|
+
)
|
|
57
|
+
if token.type != "OPERAND" or token.subtype != "RANGE":
|
|
58
|
+
continue
|
|
59
|
+
ref = _resolve(token.value, source, snapshot, set())
|
|
60
|
+
fact.references.append(ref)
|
|
61
|
+
if ref.external_workbook:
|
|
62
|
+
facts.external_references.append(
|
|
63
|
+
ExternalReferenceFact(ref.external_workbook, ref.reference, source)
|
|
64
|
+
)
|
|
65
|
+
if ref.status not in {"resolved", "external"}:
|
|
66
|
+
facts.diagnostics.append(
|
|
67
|
+
ReferenceDiagnostic(ref.status + "_reference", ref.reference, [source])
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _resolve(text, source, snapshot, visiting):
|
|
72
|
+
sheet_name, local = source.sheet_name, text
|
|
73
|
+
if "!" in text:
|
|
74
|
+
qualifier, local = text.rsplit("!", 1)
|
|
75
|
+
sheet_name = (
|
|
76
|
+
qualifier[1:-1] if qualifier.startswith("'") and qualifier.endswith("'") else qualifier
|
|
77
|
+
)
|
|
78
|
+
sheet_name = sheet_name.replace("''", "'")
|
|
79
|
+
external = re.search(r"\[([^]]+)\]", sheet_name)
|
|
80
|
+
if external:
|
|
81
|
+
return FormulaReference(text, "external", external_workbook=external.group(1))
|
|
82
|
+
if ":" in sheet_name:
|
|
83
|
+
return FormulaReference(text, "unsupported")
|
|
84
|
+
canonical = next(
|
|
85
|
+
(s.name for s in snapshot.sheets if s.name.casefold() == sheet_name.casefold()), None
|
|
86
|
+
)
|
|
87
|
+
if canonical is None:
|
|
88
|
+
return FormulaReference(text, "unresolved")
|
|
89
|
+
if _A1.fullmatch(local):
|
|
90
|
+
normalized = local.replace("$", "").upper()
|
|
91
|
+
bounds = range_boundaries(normalized)
|
|
92
|
+
if (
|
|
93
|
+
bounds[2] <= 16384
|
|
94
|
+
and bounds[3] <= 1048576
|
|
95
|
+
and bounds[0] <= bounds[2]
|
|
96
|
+
and bounds[1] <= bounds[3]
|
|
97
|
+
):
|
|
98
|
+
return FormulaReference(text, "resolved", [SourceRef(canonical, normalized)])
|
|
99
|
+
return FormulaReference(text, "unsupported")
|
|
100
|
+
names = [
|
|
101
|
+
n
|
|
102
|
+
for n in snapshot.reference_facts.defined_names
|
|
103
|
+
if n.name.casefold() == local.casefold() and n.scope_sheet in (None, canonical)
|
|
104
|
+
]
|
|
105
|
+
names.sort(key=lambda n: n.scope_sheet is None)
|
|
106
|
+
if names:
|
|
107
|
+
name = names[0]
|
|
108
|
+
key = (name.scope_sheet, name.name.casefold())
|
|
109
|
+
if key in visiting:
|
|
110
|
+
return FormulaReference(text, "circular")
|
|
111
|
+
# Names may contain constants or formulas. Only explicit range/name aliases resolve.
|
|
112
|
+
definition = name.definition.lstrip("=")
|
|
113
|
+
if re.search(r"(?i)\b(?:INDIRECT|OFFSET)\s*\(", definition):
|
|
114
|
+
return FormulaReference(text, "dynamic")
|
|
115
|
+
if re.fullmatch(r"(?:[+-]?\d+(?:\.\d+)?|TRUE|FALSE|\".*\")", definition, re.I):
|
|
116
|
+
return FormulaReference(text, "resolved")
|
|
117
|
+
resolved = _resolve(
|
|
118
|
+
definition, SourceRef(canonical, source.range), snapshot, visiting | {key}
|
|
119
|
+
)
|
|
120
|
+
return FormulaReference(text, resolved.status, resolved.targets, resolved.external_workbook)
|
|
121
|
+
if any(t.name.casefold() == local.casefold() for t in snapshot.reference_facts.tables):
|
|
122
|
+
return _structured(text, local + "[#Data]", source, snapshot)
|
|
123
|
+
if re.fullmatch(r"\$?[A-Za-z]+:\$?[A-Za-z]+|\$?\d+:\$?\d+", local):
|
|
124
|
+
return FormulaReference(text, "unsupported")
|
|
125
|
+
if "[" in local:
|
|
126
|
+
return _structured(text, local, source, snapshot)
|
|
127
|
+
return FormulaReference(text, "unresolved")
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _structured(text, local, source, snapshot):
|
|
131
|
+
table_name = local.split("[", 1)[0]
|
|
132
|
+
tables = [
|
|
133
|
+
t for t in snapshot.reference_facts.tables if t.name.casefold() == table_name.casefold()
|
|
134
|
+
]
|
|
135
|
+
if not table_name:
|
|
136
|
+
from langparse.workbooks.lineage import overlaps
|
|
137
|
+
|
|
138
|
+
tables = [t for t in snapshot.reference_facts.tables if overlaps(t.source_ref, source)]
|
|
139
|
+
if len(tables) != 1:
|
|
140
|
+
return FormulaReference(text, "unresolved")
|
|
141
|
+
table = tables[0]
|
|
142
|
+
left, top, right, bottom = range_boundaries(table.source_ref.range)
|
|
143
|
+
parts = re.findall(r"\[([^\[\]]+)\]", local)
|
|
144
|
+
selectors = [p for p in parts if p.startswith("#")]
|
|
145
|
+
columns = [p.lstrip("@") for p in parts if not p.startswith("#")]
|
|
146
|
+
if any(s not in {"#All", "#Data", "#Headers", "#Totals", "#This Row"} for s in selectors):
|
|
147
|
+
return FormulaReference(text, "unsupported")
|
|
148
|
+
if len(selectors) > 1:
|
|
149
|
+
return FormulaReference(text, "unsupported")
|
|
150
|
+
selector = selectors[0] if selectors else "#Data"
|
|
151
|
+
if selector == "#Headers":
|
|
152
|
+
bottom = top + table.header_rows - 1
|
|
153
|
+
elif selector == "#Totals":
|
|
154
|
+
top = bottom - table.totals_rows + 1
|
|
155
|
+
elif selector != "#All":
|
|
156
|
+
top += table.header_rows
|
|
157
|
+
bottom -= table.totals_rows
|
|
158
|
+
if selector == "#This Row" or "@" in local:
|
|
159
|
+
row = range_boundaries(source.range)[1]
|
|
160
|
+
if source.sheet_name != table.source_ref.sheet_name or not top <= row <= bottom:
|
|
161
|
+
return FormulaReference(text, "unsupported")
|
|
162
|
+
top = bottom = row
|
|
163
|
+
if columns:
|
|
164
|
+
indices = []
|
|
165
|
+
for column in columns:
|
|
166
|
+
matches = [
|
|
167
|
+
i for i, name in enumerate(table.columns) if name.casefold() == column.casefold()
|
|
168
|
+
]
|
|
169
|
+
if not matches:
|
|
170
|
+
return FormulaReference(text, "unresolved")
|
|
171
|
+
indices.append(left + matches[0])
|
|
172
|
+
if len(indices) > 2 or (len(indices) == 2 and ":" not in local):
|
|
173
|
+
return FormulaReference(text, "unsupported")
|
|
174
|
+
left, right = indices[0], indices[-1]
|
|
175
|
+
if bottom < top or right < left:
|
|
176
|
+
return FormulaReference(text, "unsupported")
|
|
177
|
+
target = f"{get_column_letter(left)}{top}:{get_column_letter(right)}{bottom}"
|
|
178
|
+
return FormulaReference(text, "resolved", [SourceRef(table.source_ref.sheet_name, target)])
|