langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
import shutil
|
|
7
|
+
import uuid
|
|
8
|
+
from dataclasses import asdict, dataclass, fields
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from langparse import __version__
|
|
13
|
+
from langparse.services.parse_service import ParseService
|
|
14
|
+
from langparse.workbooks.evaluation.schema import validate_output_dir_isolation
|
|
15
|
+
from langparse.workbooks.quality.evaluator import (
|
|
16
|
+
WORKBOOK_QUALITY_METRIC_SCHEMA_VERSION,
|
|
17
|
+
WorkbookQualityMetrics,
|
|
18
|
+
evaluate_workbook_result,
|
|
19
|
+
)
|
|
20
|
+
from langparse.workbooks.quality.schema import load_workbook_quality_manifest
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class WorkbookQualityReportError(RuntimeError):
|
|
24
|
+
"""Raised when a workbook quality report cannot be safely published."""
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass(frozen=True)
|
|
28
|
+
class WorkbookQualityBenchmarkReport:
|
|
29
|
+
run_digest: str
|
|
30
|
+
output_path: Path
|
|
31
|
+
summary: dict[str, Any]
|
|
32
|
+
results: tuple[dict[str, Any], ...]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class WorkbookQualityBenchmarkService:
|
|
36
|
+
"""Evaluate complete workbook results against versioned structural truth."""
|
|
37
|
+
|
|
38
|
+
def __init__(self, parse_service: ParseService | None = None) -> None:
|
|
39
|
+
self._parse_service = parse_service or ParseService()
|
|
40
|
+
|
|
41
|
+
def run(
|
|
42
|
+
self,
|
|
43
|
+
manifest_path: str | Path,
|
|
44
|
+
*,
|
|
45
|
+
output_dir: str | Path,
|
|
46
|
+
markdown: bool = True,
|
|
47
|
+
) -> WorkbookQualityBenchmarkReport:
|
|
48
|
+
manifest = load_workbook_quality_manifest(manifest_path)
|
|
49
|
+
output_root = Path(output_dir).resolve()
|
|
50
|
+
validate_output_dir_isolation(manifest.source_root, output_root)
|
|
51
|
+
|
|
52
|
+
sample_results: list[dict[str, Any]] = []
|
|
53
|
+
for sample in manifest.samples:
|
|
54
|
+
sample_path = (manifest.source_root / sample.path).resolve()
|
|
55
|
+
before = _file_identity(sample_path)
|
|
56
|
+
if before[2] != sample.sha256:
|
|
57
|
+
raise WorkbookQualityReportError(
|
|
58
|
+
f"Source workbook no longer matches manifest: {sample.sample_id}"
|
|
59
|
+
)
|
|
60
|
+
parsed = self._parse_service.parse_result(sample_path)
|
|
61
|
+
metrics = evaluate_workbook_result(sample.expectation, parsed)
|
|
62
|
+
after = _file_identity(sample_path)
|
|
63
|
+
if after != before:
|
|
64
|
+
raise WorkbookQualityReportError(
|
|
65
|
+
f"Source workbook changed during evaluation: {sample.sample_id}"
|
|
66
|
+
)
|
|
67
|
+
sample_results.append(
|
|
68
|
+
{
|
|
69
|
+
"sample_id": sample.sample_id,
|
|
70
|
+
"sha256": sample.sha256,
|
|
71
|
+
"metrics": asdict(metrics),
|
|
72
|
+
}
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
aggregate = _aggregate_metrics(sample_results)
|
|
76
|
+
gate_reasons = _gate_reasons(
|
|
77
|
+
aggregate,
|
|
78
|
+
minimum=manifest.quality_gate.minimum,
|
|
79
|
+
maximum=manifest.quality_gate.maximum,
|
|
80
|
+
)
|
|
81
|
+
run_payload = {
|
|
82
|
+
"schema_version": 1,
|
|
83
|
+
"metric_schema_version": WORKBOOK_QUALITY_METRIC_SCHEMA_VERSION,
|
|
84
|
+
"dataset_id": manifest.dataset_id,
|
|
85
|
+
"dataset_version": manifest.dataset_version,
|
|
86
|
+
"dataset_digest": manifest.dataset_digest,
|
|
87
|
+
"split": manifest.split,
|
|
88
|
+
"parser_version": __version__,
|
|
89
|
+
"quality_gate": {
|
|
90
|
+
"minimum": manifest.quality_gate.minimum,
|
|
91
|
+
"maximum": manifest.quality_gate.maximum,
|
|
92
|
+
},
|
|
93
|
+
"artifact_options": {"markdown": markdown},
|
|
94
|
+
"results": sample_results,
|
|
95
|
+
}
|
|
96
|
+
run_digest = _payload_digest(run_payload)
|
|
97
|
+
summary = {
|
|
98
|
+
"schema_version": 1,
|
|
99
|
+
"metric_schema_version": WORKBOOK_QUALITY_METRIC_SCHEMA_VERSION,
|
|
100
|
+
"dataset_id": manifest.dataset_id,
|
|
101
|
+
"dataset_version": manifest.dataset_version,
|
|
102
|
+
"dataset_digest": manifest.dataset_digest,
|
|
103
|
+
"split": manifest.split,
|
|
104
|
+
"parser_version": __version__,
|
|
105
|
+
"sample_count": len(sample_results),
|
|
106
|
+
"metrics": aggregate,
|
|
107
|
+
"quality_gate": run_payload["quality_gate"],
|
|
108
|
+
"artifact_options": run_payload["artifact_options"],
|
|
109
|
+
"gate_reasons": gate_reasons,
|
|
110
|
+
"status": "passed" if not gate_reasons else "failed",
|
|
111
|
+
"run_digest": run_digest,
|
|
112
|
+
}
|
|
113
|
+
run_dir = output_root / run_digest.removeprefix("sha256:")
|
|
114
|
+
_publish_report(run_dir, sample_results, summary, markdown=markdown)
|
|
115
|
+
return WorkbookQualityBenchmarkReport(
|
|
116
|
+
run_digest=run_digest,
|
|
117
|
+
output_path=run_dir,
|
|
118
|
+
summary=summary,
|
|
119
|
+
results=tuple(sample_results),
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _aggregate_metrics(results: list[dict[str, Any]]) -> dict[str, float | None]:
|
|
124
|
+
aggregated: dict[str, float | None] = {}
|
|
125
|
+
for metric_field in fields(WorkbookQualityMetrics):
|
|
126
|
+
values = [
|
|
127
|
+
result["metrics"][metric_field.name]
|
|
128
|
+
for result in results
|
|
129
|
+
if result["metrics"][metric_field.name] is not None
|
|
130
|
+
]
|
|
131
|
+
aggregated[metric_field.name] = sum(values) / len(values) if values else None
|
|
132
|
+
return aggregated
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _gate_reasons(
|
|
136
|
+
metrics: dict[str, float | None],
|
|
137
|
+
*,
|
|
138
|
+
minimum: dict[str, float],
|
|
139
|
+
maximum: dict[str, float],
|
|
140
|
+
) -> list[str]:
|
|
141
|
+
reasons = []
|
|
142
|
+
for name, threshold in sorted(minimum.items()):
|
|
143
|
+
value = metrics.get(name)
|
|
144
|
+
if value is None or value < threshold:
|
|
145
|
+
reasons.append(f"{name}_below_minimum")
|
|
146
|
+
for name, threshold in sorted(maximum.items()):
|
|
147
|
+
value = metrics.get(name)
|
|
148
|
+
if value is None or value > threshold:
|
|
149
|
+
reasons.append(f"{name}_above_maximum")
|
|
150
|
+
return reasons
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _payload_digest(payload: dict[str, Any]) -> str:
|
|
154
|
+
encoded = json.dumps(payload, sort_keys=True, separators=(",", ":")).encode("utf-8")
|
|
155
|
+
return f"sha256:{hashlib.sha256(encoded).hexdigest()}"
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _file_identity(path: Path) -> tuple[int, int, str]:
|
|
159
|
+
stat = path.stat()
|
|
160
|
+
digest = f"sha256:{hashlib.sha256(path.read_bytes()).hexdigest()}"
|
|
161
|
+
return stat.st_size, stat.st_mtime_ns, digest
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _publish_report(
|
|
165
|
+
run_dir: Path,
|
|
166
|
+
results: list[dict[str, Any]],
|
|
167
|
+
summary: dict[str, Any],
|
|
168
|
+
*,
|
|
169
|
+
markdown: bool,
|
|
170
|
+
) -> None:
|
|
171
|
+
run_dir.parent.mkdir(parents=True, exist_ok=True)
|
|
172
|
+
temporary = run_dir.parent / f".{run_dir.name}.{uuid.uuid4().hex}.tmp"
|
|
173
|
+
temporary.mkdir()
|
|
174
|
+
try:
|
|
175
|
+
results_text = "".join(
|
|
176
|
+
json.dumps(result, ensure_ascii=False, sort_keys=True) + "\n" for result in results
|
|
177
|
+
)
|
|
178
|
+
(temporary / "workbook-quality-results.jsonl").write_text(
|
|
179
|
+
results_text,
|
|
180
|
+
encoding="utf-8",
|
|
181
|
+
)
|
|
182
|
+
(temporary / "workbook-quality-summary.json").write_text(
|
|
183
|
+
json.dumps(summary, ensure_ascii=False, indent=2, sort_keys=True) + "\n",
|
|
184
|
+
encoding="utf-8",
|
|
185
|
+
)
|
|
186
|
+
if markdown:
|
|
187
|
+
(temporary / "workbook-quality-summary.md").write_text(
|
|
188
|
+
_summary_markdown(summary),
|
|
189
|
+
encoding="utf-8",
|
|
190
|
+
)
|
|
191
|
+
if run_dir.exists():
|
|
192
|
+
if not _directories_equal(run_dir, temporary):
|
|
193
|
+
raise WorkbookQualityReportError(
|
|
194
|
+
f"Immutable report collision for {summary['run_digest']}"
|
|
195
|
+
)
|
|
196
|
+
return
|
|
197
|
+
os.replace(temporary, run_dir)
|
|
198
|
+
finally:
|
|
199
|
+
if temporary.exists():
|
|
200
|
+
shutil.rmtree(temporary)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _directories_equal(left: Path, right: Path) -> bool:
|
|
204
|
+
left_files = {path.relative_to(left): path for path in left.rglob("*") if path.is_file()}
|
|
205
|
+
right_files = {path.relative_to(right): path for path in right.rglob("*") if path.is_file()}
|
|
206
|
+
return left_files.keys() == right_files.keys() and all(
|
|
207
|
+
left_files[name].read_bytes() == right_files[name].read_bytes() for name in left_files
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _summary_markdown(summary: dict[str, Any]) -> str:
|
|
212
|
+
lines = [
|
|
213
|
+
f"# Workbook Quality Summary ({summary['dataset_id']})",
|
|
214
|
+
"",
|
|
215
|
+
f"- **Dataset Version**: {summary['dataset_version']}",
|
|
216
|
+
f"- **Dataset Digest**: {summary['dataset_digest']}",
|
|
217
|
+
f"- **Split**: {summary['split']}",
|
|
218
|
+
f"- **Parser Version**: {summary['parser_version']}",
|
|
219
|
+
f"- **Metric Schema Version**: {summary['metric_schema_version']}",
|
|
220
|
+
f"- **Status**: {summary['status']}",
|
|
221
|
+
f"- **Run Digest**: {summary['run_digest']}",
|
|
222
|
+
f"- **Gate Reasons**: {', '.join(summary['gate_reasons']) or 'none'}",
|
|
223
|
+
"",
|
|
224
|
+
"## Metrics",
|
|
225
|
+
"",
|
|
226
|
+
"| Metric | Value |",
|
|
227
|
+
"| :--- | :--- |",
|
|
228
|
+
]
|
|
229
|
+
lines.extend(f"| {name} | {value} |" for name, value in summary["metrics"].items())
|
|
230
|
+
return "\n".join(lines) + "\n"
|
langparse/types.py
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
StructuredData = dict[str, Any]
|
|
7
|
+
BoundingBox = list[float] | None
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass
|
|
11
|
+
class Chunk:
|
|
12
|
+
"""
|
|
13
|
+
Represents a chunk of text derived from a document.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
content: str
|
|
17
|
+
metadata: dict[str, Any] = field(default_factory=dict)
|
|
18
|
+
structured_payload: StructuredData = field(default_factory=dict)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class Document:
|
|
23
|
+
"""
|
|
24
|
+
Represents a parsed document.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
content: str # The full text content (usually Markdown)
|
|
28
|
+
metadata: dict[str, Any] = field(default_factory=dict)
|
|
29
|
+
chunks: list[Chunk] = field(default_factory=list)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class ParsedElement:
|
|
34
|
+
kind: str
|
|
35
|
+
text: str = ""
|
|
36
|
+
bbox: BoundingBox = None
|
|
37
|
+
metadata: StructuredData = field(default_factory=dict)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass
|
|
41
|
+
class ParsedPageResult:
|
|
42
|
+
"""
|
|
43
|
+
Normalized parsed page result stored in the final document model.
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
page_number: int
|
|
47
|
+
markdown_content: str
|
|
48
|
+
plain_text: str = ""
|
|
49
|
+
elements: list[ParsedElement] = field(default_factory=list)
|
|
50
|
+
tables: list[StructuredData] = field(default_factory=list)
|
|
51
|
+
images: list[StructuredData] = field(default_factory=list)
|
|
52
|
+
metadata: StructuredData = field(default_factory=dict)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclass
|
|
56
|
+
class ParsedStructure:
|
|
57
|
+
"""Typed structural representation produced by a format parser."""
|
|
58
|
+
|
|
59
|
+
kind: str
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@dataclass
|
|
63
|
+
class ParseDiagnostics:
|
|
64
|
+
"""Machine-readable quality and coverage information for a parse."""
|
|
65
|
+
|
|
66
|
+
status: str = "success"
|
|
67
|
+
coverage_ratio: float = 1.0
|
|
68
|
+
reconstruction_passed: bool = True
|
|
69
|
+
source_ref_validity_ratio: float = 1.0
|
|
70
|
+
block_count_by_kind: dict[str, int] = field(default_factory=dict)
|
|
71
|
+
ambiguous_regions: list[StructuredData] = field(default_factory=list)
|
|
72
|
+
model_calls: list[StructuredData] = field(default_factory=list)
|
|
73
|
+
unsupported_features: list[str] = field(default_factory=list)
|
|
74
|
+
warnings: list[str] = field(default_factory=list)
|
|
75
|
+
errors: list[str] = field(default_factory=list)
|
|
76
|
+
timings_by_stage: dict[str, float] = field(default_factory=dict)
|
|
77
|
+
continuation_candidates: list[StructuredData] = field(default_factory=list)
|
|
78
|
+
region_diagnostics: list[StructuredData] = field(default_factory=list)
|
|
79
|
+
object_coverage_ratio: float | None = 1.0
|
|
80
|
+
reference_diagnostics: list[StructuredData] = field(default_factory=list)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@dataclass
|
|
84
|
+
class ParsedDocumentResult:
|
|
85
|
+
source: str
|
|
86
|
+
filename: str
|
|
87
|
+
engine: str
|
|
88
|
+
pages: list[ParsedPageResult] = field(default_factory=list)
|
|
89
|
+
markdown_content: str = ""
|
|
90
|
+
metadata: StructuredData = field(default_factory=dict)
|
|
91
|
+
#: Whether page numbers are real boundaries. Flow formats without intrinsic
|
|
92
|
+
#: pagination (plain Markdown) set this False so downstream code neither
|
|
93
|
+
#: injects page markers nor scores page coverage against a fiction.
|
|
94
|
+
paginated: bool = True
|
|
95
|
+
structure: ParsedStructure | None = None
|
|
96
|
+
chunks: list[Chunk] = field(default_factory=list)
|
|
97
|
+
diagnostics: ParseDiagnostics | None = None
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""Lossless workbook facts and structural intermediate representation.
|
|
2
|
+
|
|
3
|
+
Workbook algorithms depend on the optional Excel stack. Keep those imports
|
|
4
|
+
lazy so the dependency-free core package and CLI discovery remain usable until
|
|
5
|
+
a caller actually selects Excel functionality.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from importlib import import_module
|
|
9
|
+
|
|
10
|
+
from langparse.workbooks.types import (
|
|
11
|
+
CandidateRegion,
|
|
12
|
+
CellSnapshot,
|
|
13
|
+
FormBlock,
|
|
14
|
+
FormField,
|
|
15
|
+
HeaderColumn,
|
|
16
|
+
LogicalRow,
|
|
17
|
+
LogicalTable,
|
|
18
|
+
MatrixBlock,
|
|
19
|
+
MatrixHeader,
|
|
20
|
+
RegionAnchor,
|
|
21
|
+
SheetIR,
|
|
22
|
+
SheetSnapshot,
|
|
23
|
+
SourceRef,
|
|
24
|
+
TableContinuation,
|
|
25
|
+
TableFragment,
|
|
26
|
+
TableSection,
|
|
27
|
+
TextBlock,
|
|
28
|
+
TextLine,
|
|
29
|
+
WorkbookBlock,
|
|
30
|
+
WorkbookIR,
|
|
31
|
+
WorkbookSnapshot,
|
|
32
|
+
stable_id,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
_LAZY_EXPORTS = {
|
|
36
|
+
"WorkbookBundle": ("langparse.workbooks.bundle", "WorkbookBundle"),
|
|
37
|
+
"WorkbookLineage": ("langparse.workbooks.lineage", "WorkbookLineage"),
|
|
38
|
+
"OOXMLWorkbookAdapter": ("langparse.workbooks.adapters", "OOXMLWorkbookAdapter"),
|
|
39
|
+
"WorkbookAdapter": ("langparse.workbooks.adapters", "WorkbookAdapter"),
|
|
40
|
+
"assemble_baseline": ("langparse.workbooks.assembly", "assemble_baseline"),
|
|
41
|
+
"assemble_workbook": ("langparse.workbooks.assembly", "assemble_workbook"),
|
|
42
|
+
"validate_workbook_source_refs": (
|
|
43
|
+
"langparse.workbooks.assembly",
|
|
44
|
+
"validate_workbook_source_refs",
|
|
45
|
+
),
|
|
46
|
+
"interpret_form_block": ("langparse.workbooks.blocks", "interpret_form_block"),
|
|
47
|
+
"interpret_matrix_block": ("langparse.workbooks.blocks", "interpret_matrix_block"),
|
|
48
|
+
"interpret_text_block": ("langparse.workbooks.blocks", "interpret_text_block"),
|
|
49
|
+
"detect_candidate_regions": ("langparse.workbooks.regions", "detect_candidate_regions"),
|
|
50
|
+
"compatibility_pages": ("langparse.workbooks.rendering", "compatibility_pages"),
|
|
51
|
+
"render_workbook_markdown": ("langparse.workbooks.rendering", "render_workbook_markdown"),
|
|
52
|
+
"interpret_logical_table": ("langparse.workbooks.tables", "interpret_logical_table"),
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def __getattr__(name: str):
|
|
57
|
+
target = _LAZY_EXPORTS.get(name)
|
|
58
|
+
if target is None:
|
|
59
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
60
|
+
module_name, attribute_name = target
|
|
61
|
+
value = getattr(import_module(module_name), attribute_name)
|
|
62
|
+
globals()[name] = value
|
|
63
|
+
return value
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
__all__ = [
|
|
67
|
+
"WorkbookBundle",
|
|
68
|
+
"WorkbookLineage",
|
|
69
|
+
"CandidateRegion",
|
|
70
|
+
"CellSnapshot",
|
|
71
|
+
"FormBlock",
|
|
72
|
+
"FormField",
|
|
73
|
+
"HeaderColumn",
|
|
74
|
+
"LogicalRow",
|
|
75
|
+
"LogicalTable",
|
|
76
|
+
"MatrixBlock",
|
|
77
|
+
"MatrixHeader",
|
|
78
|
+
"OOXMLWorkbookAdapter",
|
|
79
|
+
"RegionAnchor",
|
|
80
|
+
"SheetIR",
|
|
81
|
+
"SheetSnapshot",
|
|
82
|
+
"SourceRef",
|
|
83
|
+
"TableFragment",
|
|
84
|
+
"TableSection",
|
|
85
|
+
"TableContinuation",
|
|
86
|
+
"TextBlock",
|
|
87
|
+
"TextLine",
|
|
88
|
+
"WorkbookBlock",
|
|
89
|
+
"WorkbookAdapter",
|
|
90
|
+
"WorkbookIR",
|
|
91
|
+
"WorkbookSnapshot",
|
|
92
|
+
"assemble_baseline",
|
|
93
|
+
"assemble_workbook",
|
|
94
|
+
"compatibility_pages",
|
|
95
|
+
"detect_candidate_regions",
|
|
96
|
+
"interpret_logical_table",
|
|
97
|
+
"interpret_form_block",
|
|
98
|
+
"interpret_matrix_block",
|
|
99
|
+
"interpret_text_block",
|
|
100
|
+
"render_workbook_markdown",
|
|
101
|
+
"stable_id",
|
|
102
|
+
"validate_workbook_source_refs",
|
|
103
|
+
]
|