langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections import Counter
|
|
4
|
+
from collections.abc import Iterable
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def output_filename(source, fmt: str) -> str:
|
|
9
|
+
suffix = ".md" if fmt == "markdown" else ".json"
|
|
10
|
+
return f"{Path(source).stem}{suffix}"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def extension_tagged_filename(source, fmt: str) -> str:
|
|
14
|
+
"""``report.docx`` -> ``report-docx.md``, to tell same-stem siblings apart."""
|
|
15
|
+
source_path = Path(source)
|
|
16
|
+
suffix = ".md" if fmt == "markdown" else ".json"
|
|
17
|
+
source_kind = source_path.suffix.lower().lstrip(".")
|
|
18
|
+
stem = f"{source_path.stem}-{source_kind}" if source_kind else source_path.stem
|
|
19
|
+
return f"{stem}{suffix}"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def resolve_output_path(
|
|
23
|
+
source,
|
|
24
|
+
fmt: str,
|
|
25
|
+
used_paths: set[Path],
|
|
26
|
+
preferred_filename: str | None = None,
|
|
27
|
+
) -> Path:
|
|
28
|
+
"""
|
|
29
|
+
Pick a repo-relative output path for one source, widening the parent prefix
|
|
30
|
+
until it no longer collides with a path already handed out.
|
|
31
|
+
|
|
32
|
+
Two sources sharing a stem (``alpha/report.pdf`` and ``beta/report.pdf``)
|
|
33
|
+
must not both resolve to ``report.md`` — the second would silently
|
|
34
|
+
overwrite the first.
|
|
35
|
+
"""
|
|
36
|
+
source_path = Path(source)
|
|
37
|
+
filename = preferred_filename or output_filename(source_path, fmt)
|
|
38
|
+
parent_parts = [
|
|
39
|
+
part for part in source_path.parent.parts if part not in {"", ".", source_path.anchor}
|
|
40
|
+
]
|
|
41
|
+
|
|
42
|
+
def with_prefix(width: int) -> Path:
|
|
43
|
+
return Path(*parent_parts[-width:]) / filename if width else Path(filename)
|
|
44
|
+
|
|
45
|
+
if source_path.is_absolute():
|
|
46
|
+
widths: Iterable[int] = range(0, len(parent_parts) + 1)
|
|
47
|
+
else:
|
|
48
|
+
widths = [len(parent_parts), *range(1, len(parent_parts)), 0]
|
|
49
|
+
|
|
50
|
+
for width in widths:
|
|
51
|
+
candidate = with_prefix(width)
|
|
52
|
+
if candidate not in used_paths:
|
|
53
|
+
used_paths.add(candidate)
|
|
54
|
+
return candidate
|
|
55
|
+
|
|
56
|
+
stem, suffix = Path(filename).stem, Path(filename).suffix
|
|
57
|
+
counter = 1
|
|
58
|
+
while True:
|
|
59
|
+
candidate = Path(f"{stem}-{counter}{suffix}")
|
|
60
|
+
if candidate not in used_paths:
|
|
61
|
+
used_paths.add(candidate)
|
|
62
|
+
return candidate
|
|
63
|
+
counter += 1
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def resolve_output_paths(sources: Iterable, fmt: str) -> list[Path]:
|
|
67
|
+
"""
|
|
68
|
+
Resolve every source to a distinct relative output path, in input order.
|
|
69
|
+
|
|
70
|
+
Collisions get the disambiguator that actually carries information:
|
|
71
|
+
same-stem siblings in one directory (``report.pdf`` next to ``report.docx``)
|
|
72
|
+
are told apart by source format, because widening the parent prefix would
|
|
73
|
+
only scatter siblings across unrelated output directories. Same-stem sources
|
|
74
|
+
in *different* directories keep the plain name and widen the prefix instead.
|
|
75
|
+
"""
|
|
76
|
+
sources = list(sources)
|
|
77
|
+
sibling_counts = Counter((Path(source).parent, Path(source).stem) for source in sources)
|
|
78
|
+
|
|
79
|
+
used_paths: set[Path] = set()
|
|
80
|
+
resolved: list[Path] = []
|
|
81
|
+
for source in sources:
|
|
82
|
+
source_path = Path(source)
|
|
83
|
+
has_same_dir_twin = sibling_counts[(source_path.parent, source_path.stem)] > 1
|
|
84
|
+
preferred = extension_tagged_filename(source, fmt) if has_same_dir_twin else None
|
|
85
|
+
resolved.append(resolve_output_path(source, fmt, used_paths, preferred))
|
|
86
|
+
return resolved
|
|
@@ -0,0 +1,523 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from collections.abc import Iterable, Iterator
|
|
5
|
+
from dataclasses import asdict
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from langparse.chunkers.profiles import (
|
|
9
|
+
ChunkProfileNotSupportedError,
|
|
10
|
+
resolve_workbook_chunk_policy,
|
|
11
|
+
)
|
|
12
|
+
from langparse.chunkers.registry import create_chunker
|
|
13
|
+
from langparse.config import settings
|
|
14
|
+
from langparse.core.rendering import document_from_result
|
|
15
|
+
from langparse.engines.pdf.deepdoc_engine import DeepDocEngine
|
|
16
|
+
from langparse.engines.pdf.mineru import MinerUEngine
|
|
17
|
+
from langparse.engines.pdf.other import PaddleOCRVLEngine
|
|
18
|
+
from langparse.engines.pdf.simple import SimplePDFEngine
|
|
19
|
+
from langparse.engines.pdf.vision_llm import VisionLLMEngine
|
|
20
|
+
from langparse.parsers.registry import (
|
|
21
|
+
is_supported,
|
|
22
|
+
parser_kind_for,
|
|
23
|
+
unsupported_extension_error,
|
|
24
|
+
)
|
|
25
|
+
from langparse.progress import ProgressCallback, ProgressReporter
|
|
26
|
+
from langparse.services.output_paths import (
|
|
27
|
+
output_filename,
|
|
28
|
+
resolve_output_path,
|
|
29
|
+
resolve_output_paths,
|
|
30
|
+
)
|
|
31
|
+
from langparse.types import (
|
|
32
|
+
Chunk,
|
|
33
|
+
Document,
|
|
34
|
+
ParsedDocumentResult,
|
|
35
|
+
ParseDiagnostics,
|
|
36
|
+
ParsedPageResult,
|
|
37
|
+
)
|
|
38
|
+
from langparse.workbooks.modeling import WorkbookDisambiguation
|
|
39
|
+
from langparse.workbooks.types import WorkbookIR
|
|
40
|
+
|
|
41
|
+
#: Engines a caller can actually select. Advertising an engine that raises
|
|
42
|
+
#: NotImplementedError only once it runs wastes the user's configuration effort
|
|
43
|
+
#: and, for MinerU-scale setups, their model downloads.
|
|
44
|
+
ENGINE_MAP = {
|
|
45
|
+
"simple": SimplePDFEngine,
|
|
46
|
+
"mineru": MinerUEngine,
|
|
47
|
+
"deepdoc": DeepDocEngine,
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
#: Reserved names with adapters in the tree but no working implementation.
|
|
51
|
+
#: Selecting one fails immediately with an explanation instead of at parse time.
|
|
52
|
+
PLANNED_ENGINES = {
|
|
53
|
+
"vision_llm": VisionLLMEngine,
|
|
54
|
+
"paddle": PaddleOCRVLEngine,
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class ParseService:
|
|
59
|
+
def chunk_result(
|
|
60
|
+
self,
|
|
61
|
+
parsed: ParsedDocumentResult,
|
|
62
|
+
chunker=None,
|
|
63
|
+
*,
|
|
64
|
+
chunk_profile: str | None = None,
|
|
65
|
+
chunk_strategy: str | None = None,
|
|
66
|
+
chunk_options: dict | None = None,
|
|
67
|
+
) -> list[Chunk]:
|
|
68
|
+
"""Chunk a parse result from its richest available representation."""
|
|
69
|
+
if chunker is not None and (chunk_strategy is not None or chunk_options is not None):
|
|
70
|
+
raise ValueError("custom chunker and chunk_strategy/options are mutually exclusive")
|
|
71
|
+
if chunker is not None and chunk_profile is not None:
|
|
72
|
+
raise ValueError("custom chunker and chunk_profile are mutually exclusive")
|
|
73
|
+
|
|
74
|
+
if chunker is not None:
|
|
75
|
+
if parsed.structure is not None and parsed.structure.kind == "workbook":
|
|
76
|
+
return chunker.chunk(parsed)
|
|
77
|
+
return chunker.chunk(document_from_result(parsed))
|
|
78
|
+
|
|
79
|
+
selected = create_chunker(chunk_strategy or "semantic", **(chunk_options or {}))
|
|
80
|
+
policy = resolve_workbook_chunk_policy(chunk_profile)
|
|
81
|
+
if isinstance(parsed.structure, WorkbookIR):
|
|
82
|
+
from langparse.chunkers.workbook import WorkbookStructuralChunker
|
|
83
|
+
|
|
84
|
+
return WorkbookStructuralChunker(profile=policy.name).chunk(parsed)
|
|
85
|
+
|
|
86
|
+
if policy.name.value == "analysis":
|
|
87
|
+
raise ChunkProfileNotSupportedError("analysis chunk profile requires WorkbookIR")
|
|
88
|
+
|
|
89
|
+
chunks = selected.chunk(document_from_result(parsed))
|
|
90
|
+
for chunk in chunks:
|
|
91
|
+
chunk.metadata["chunk_strategy"] = chunk_strategy or "semantic"
|
|
92
|
+
chunk.metadata["chunk_profile"] = policy.name.value
|
|
93
|
+
chunk.metadata["chunk_profile_version"] = policy.version
|
|
94
|
+
return chunks
|
|
95
|
+
|
|
96
|
+
def render_output(
|
|
97
|
+
self,
|
|
98
|
+
parsed: ParsedDocumentResult,
|
|
99
|
+
fmt: str,
|
|
100
|
+
chunks: list[Chunk] | None = None,
|
|
101
|
+
) -> str:
|
|
102
|
+
if fmt == "markdown":
|
|
103
|
+
if chunks is None:
|
|
104
|
+
return parsed.markdown_content
|
|
105
|
+
if not chunks:
|
|
106
|
+
return parsed.markdown_content
|
|
107
|
+
return "\n\n---\n\n".join(chunk.content for chunk in chunks)
|
|
108
|
+
if fmt == "workbook-json":
|
|
109
|
+
from langparse.workbooks.bundle import WorkbookBundle
|
|
110
|
+
|
|
111
|
+
return WorkbookBundle.from_result(parsed).to_json()
|
|
112
|
+
if fmt == "json":
|
|
113
|
+
payload = asdict(parsed)
|
|
114
|
+
if chunks is not None:
|
|
115
|
+
payload["chunks"] = [asdict(chunk) for chunk in chunks]
|
|
116
|
+
return json.dumps(payload, ensure_ascii=False, indent=2, default=_json_scalar)
|
|
117
|
+
raise ValueError(f"Unsupported output format: {fmt}")
|
|
118
|
+
|
|
119
|
+
def parse_output(
|
|
120
|
+
self,
|
|
121
|
+
file_path,
|
|
122
|
+
engine_name="simple",
|
|
123
|
+
fmt="markdown",
|
|
124
|
+
engine=None,
|
|
125
|
+
chunk=False,
|
|
126
|
+
chunk_profile: str | None = None,
|
|
127
|
+
chunk_strategy: str | None = None,
|
|
128
|
+
chunk_options: dict | None = None,
|
|
129
|
+
workbook_disambiguation: WorkbookDisambiguation | None = None,
|
|
130
|
+
**kwargs,
|
|
131
|
+
) -> str:
|
|
132
|
+
parsed = self.parse_result(
|
|
133
|
+
file_path,
|
|
134
|
+
engine_name=engine_name,
|
|
135
|
+
engine=engine,
|
|
136
|
+
chunk=chunk,
|
|
137
|
+
chunk_profile=chunk_profile,
|
|
138
|
+
chunk_strategy=chunk_strategy,
|
|
139
|
+
chunk_options=chunk_options,
|
|
140
|
+
workbook_disambiguation=workbook_disambiguation,
|
|
141
|
+
**kwargs,
|
|
142
|
+
)
|
|
143
|
+
return self.render_output(parsed, fmt, chunks=parsed.chunks if chunk else None)
|
|
144
|
+
|
|
145
|
+
def parse_batch_outputs(
|
|
146
|
+
self,
|
|
147
|
+
inputs,
|
|
148
|
+
engine_name="simple",
|
|
149
|
+
fmt="markdown",
|
|
150
|
+
engine=None,
|
|
151
|
+
chunk=False,
|
|
152
|
+
chunk_profile: str | None = None,
|
|
153
|
+
chunk_strategy: str | None = None,
|
|
154
|
+
chunk_options: dict | None = None,
|
|
155
|
+
workbook_disambiguation: WorkbookDisambiguation | None = None,
|
|
156
|
+
progress_callback: ProgressCallback | None = None,
|
|
157
|
+
**kwargs,
|
|
158
|
+
) -> list[tuple[Path, str]]:
|
|
159
|
+
# `chunk` is named explicitly rather than left in **kwargs: kwargs also
|
|
160
|
+
# feed engine construction, and MinerU folds unknown kwargs into
|
|
161
|
+
# extra_options and sends them to its API as form fields.
|
|
162
|
+
model = kwargs.pop("model", None)
|
|
163
|
+
api_key = kwargs.pop("api_key", None)
|
|
164
|
+
base_url = kwargs.pop("base_url", None)
|
|
165
|
+
kwargs.pop("disambiguation", None)
|
|
166
|
+
outputs = []
|
|
167
|
+
active_engine = engine or self._create_engine(engine_name, **kwargs)
|
|
168
|
+
for file_path in self.expand_inputs(inputs):
|
|
169
|
+
outputs.append(
|
|
170
|
+
(
|
|
171
|
+
file_path,
|
|
172
|
+
self.parse_output(
|
|
173
|
+
file_path,
|
|
174
|
+
engine_name=engine_name,
|
|
175
|
+
fmt=fmt,
|
|
176
|
+
engine=active_engine,
|
|
177
|
+
chunk=chunk,
|
|
178
|
+
chunk_profile=chunk_profile,
|
|
179
|
+
chunk_strategy=chunk_strategy,
|
|
180
|
+
chunk_options=chunk_options,
|
|
181
|
+
workbook_disambiguation=workbook_disambiguation,
|
|
182
|
+
**(
|
|
183
|
+
{"progress_callback": progress_callback}
|
|
184
|
+
if progress_callback is not None
|
|
185
|
+
else {}
|
|
186
|
+
),
|
|
187
|
+
model=model,
|
|
188
|
+
api_key=api_key,
|
|
189
|
+
base_url=base_url,
|
|
190
|
+
**kwargs,
|
|
191
|
+
),
|
|
192
|
+
)
|
|
193
|
+
)
|
|
194
|
+
return outputs
|
|
195
|
+
|
|
196
|
+
def write_output(self, content: str, output_path) -> Path:
|
|
197
|
+
destination = Path(output_path)
|
|
198
|
+
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
199
|
+
destination.write_text(content, encoding="utf-8")
|
|
200
|
+
return destination
|
|
201
|
+
|
|
202
|
+
def write_batch_outputs(self, outputs, output_dir, fmt: str) -> list[Path]:
|
|
203
|
+
output_dir = Path(output_dir)
|
|
204
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
205
|
+
|
|
206
|
+
outputs = list(outputs)
|
|
207
|
+
# Resolve all destinations together so same-stem siblings are grouped
|
|
208
|
+
# the same way BatchParseService groups them.
|
|
209
|
+
relative_paths = resolve_output_paths([source for source, _ in outputs], fmt)
|
|
210
|
+
|
|
211
|
+
written_paths = []
|
|
212
|
+
for (_, content), relative in zip(outputs, relative_paths, strict=True):
|
|
213
|
+
destination = output_dir / relative
|
|
214
|
+
self.write_output(content, destination)
|
|
215
|
+
written_paths.append(destination)
|
|
216
|
+
return written_paths
|
|
217
|
+
|
|
218
|
+
def expand_inputs(self, inputs):
|
|
219
|
+
paths = []
|
|
220
|
+
for item in self._flatten_inputs(inputs):
|
|
221
|
+
path = Path(item)
|
|
222
|
+
if not path.exists():
|
|
223
|
+
raise FileNotFoundError(f"File not found: {path}")
|
|
224
|
+
if path.is_dir():
|
|
225
|
+
paths.extend(
|
|
226
|
+
sorted(
|
|
227
|
+
child for child in path.iterdir() if child.is_file() and is_supported(child)
|
|
228
|
+
)
|
|
229
|
+
)
|
|
230
|
+
else:
|
|
231
|
+
paths.append(path)
|
|
232
|
+
return paths
|
|
233
|
+
|
|
234
|
+
def parse_result(
|
|
235
|
+
self,
|
|
236
|
+
file_path,
|
|
237
|
+
engine_name="simple",
|
|
238
|
+
engine=None,
|
|
239
|
+
chunk=False,
|
|
240
|
+
chunk_profile: str | None = None,
|
|
241
|
+
chunk_strategy: str | None = None,
|
|
242
|
+
chunk_options: dict | None = None,
|
|
243
|
+
workbook_disambiguation: str | WorkbookDisambiguation | None = None,
|
|
244
|
+
model: str | None = None,
|
|
245
|
+
api_key: str | None = None,
|
|
246
|
+
base_url: str | None = None,
|
|
247
|
+
progress_callback: ProgressCallback | None = None,
|
|
248
|
+
**kwargs,
|
|
249
|
+
):
|
|
250
|
+
"""
|
|
251
|
+
Parse any supported format into a ParsedDocumentResult.
|
|
252
|
+
|
|
253
|
+
This is the one place extension routing happens; everything else reads
|
|
254
|
+
the mapping from `langparse.parsers.registry`.
|
|
255
|
+
"""
|
|
256
|
+
path = Path(file_path)
|
|
257
|
+
reporter = ProgressReporter(path, progress_callback)
|
|
258
|
+
with reporter.operation("file"):
|
|
259
|
+
if chunk:
|
|
260
|
+
resolve_workbook_chunk_policy(chunk_profile)
|
|
261
|
+
create_chunker(chunk_strategy or "semantic", **(chunk_options or {}))
|
|
262
|
+
if not path.exists():
|
|
263
|
+
raise FileNotFoundError(f"File not found: {path}")
|
|
264
|
+
kind = parser_kind_for(path)
|
|
265
|
+
if kind is None:
|
|
266
|
+
raise unsupported_extension_error(path)
|
|
267
|
+
reporter.emit("parsing")
|
|
268
|
+
if progress_callback is not None:
|
|
269
|
+
kwargs["progress_callback"] = progress_callback
|
|
270
|
+
if kind == "pdf":
|
|
271
|
+
parsed = self._collect_pdf_document_result(
|
|
272
|
+
path,
|
|
273
|
+
engine_name=engine_name,
|
|
274
|
+
engine=engine,
|
|
275
|
+
**kwargs,
|
|
276
|
+
)
|
|
277
|
+
else:
|
|
278
|
+
parsed = self._parser_for_kind(
|
|
279
|
+
kind,
|
|
280
|
+
workbook_disambiguation,
|
|
281
|
+
model=model,
|
|
282
|
+
api_key=api_key,
|
|
283
|
+
base_url=base_url,
|
|
284
|
+
).parse_result(path, **kwargs)
|
|
285
|
+
if chunk:
|
|
286
|
+
reporter.emit("chunking")
|
|
287
|
+
self._populate_chunks(parsed, chunk_profile, chunk_strategy, chunk_options)
|
|
288
|
+
return parsed
|
|
289
|
+
|
|
290
|
+
def _populate_chunks(
|
|
291
|
+
self,
|
|
292
|
+
parsed: ParsedDocumentResult,
|
|
293
|
+
chunk_profile: str | None,
|
|
294
|
+
chunk_strategy: str | None = None,
|
|
295
|
+
chunk_options: dict | None = None,
|
|
296
|
+
) -> None:
|
|
297
|
+
policy = resolve_workbook_chunk_policy(chunk_profile)
|
|
298
|
+
try:
|
|
299
|
+
parsed.chunks = self.chunk_result(
|
|
300
|
+
parsed,
|
|
301
|
+
chunk_profile=policy.name.value,
|
|
302
|
+
chunk_strategy=chunk_strategy,
|
|
303
|
+
chunk_options=chunk_options,
|
|
304
|
+
)
|
|
305
|
+
except ChunkProfileNotSupportedError:
|
|
306
|
+
if parsed.diagnostics is None:
|
|
307
|
+
parsed.diagnostics = ParseDiagnostics()
|
|
308
|
+
if parsed.diagnostics.status != "failed":
|
|
309
|
+
parsed.diagnostics.status = "partial"
|
|
310
|
+
parsed.diagnostics.unsupported_features.append(
|
|
311
|
+
f"Chunking profile '{policy.name.value}' is not supported for engine "
|
|
312
|
+
f"'{parsed.engine}'."
|
|
313
|
+
)
|
|
314
|
+
parsed.chunks = []
|
|
315
|
+
except Exception as exc: # noqa: BLE001 - preserve parsed result at chunk boundary
|
|
316
|
+
if parsed.diagnostics is None:
|
|
317
|
+
parsed.diagnostics = ParseDiagnostics()
|
|
318
|
+
if parsed.diagnostics.status != "failed":
|
|
319
|
+
parsed.diagnostics.status = "partial"
|
|
320
|
+
parsed.diagnostics.errors.append(
|
|
321
|
+
f"Chunking profile '{policy.name.value}' failed ({type(exc).__name__})."
|
|
322
|
+
)
|
|
323
|
+
parsed.chunks = []
|
|
324
|
+
|
|
325
|
+
def _parser_for_kind(
|
|
326
|
+
self,
|
|
327
|
+
kind: str,
|
|
328
|
+
workbook_disambiguation: str | WorkbookDisambiguation | None = None,
|
|
329
|
+
*,
|
|
330
|
+
model: str | None = None,
|
|
331
|
+
api_key: str | None = None,
|
|
332
|
+
base_url: str | None = None,
|
|
333
|
+
):
|
|
334
|
+
if kind == "docx":
|
|
335
|
+
from langparse.parsers.docx_parser import DocxParser
|
|
336
|
+
|
|
337
|
+
return DocxParser()
|
|
338
|
+
if kind == "excel":
|
|
339
|
+
from langparse.parsers.excel_parser import ExcelParser
|
|
340
|
+
|
|
341
|
+
return ExcelParser(
|
|
342
|
+
disambiguation=workbook_disambiguation,
|
|
343
|
+
model=model,
|
|
344
|
+
api_key=api_key,
|
|
345
|
+
base_url=base_url,
|
|
346
|
+
)
|
|
347
|
+
if kind == "markdown":
|
|
348
|
+
from langparse.parsers.markdown_parser import MarkdownParser
|
|
349
|
+
|
|
350
|
+
return MarkdownParser()
|
|
351
|
+
raise ValueError(f"No parser registered for kind: {kind}")
|
|
352
|
+
|
|
353
|
+
def parse_file(self, file_path, engine_name="simple", engine=None, **kwargs):
|
|
354
|
+
parsed = self.parse_result(
|
|
355
|
+
file_path,
|
|
356
|
+
engine_name=engine_name,
|
|
357
|
+
engine=engine,
|
|
358
|
+
**kwargs,
|
|
359
|
+
)
|
|
360
|
+
return self._build_document_from_result(parsed)
|
|
361
|
+
|
|
362
|
+
def parse_pdf_document(self, file_path, engine_name="simple", engine=None, **kwargs):
|
|
363
|
+
return self.parse_file(file_path, engine_name=engine_name, engine=engine, **kwargs)
|
|
364
|
+
|
|
365
|
+
def parse_batch(self, inputs, engine_name="simple", engine=None, **kwargs):
|
|
366
|
+
progress_callback = kwargs.pop("progress_callback", None)
|
|
367
|
+
workbook_disambiguation = kwargs.pop("workbook_disambiguation", None)
|
|
368
|
+
model = kwargs.pop("model", None)
|
|
369
|
+
api_key = kwargs.pop("api_key", None)
|
|
370
|
+
base_url = kwargs.pop("base_url", None)
|
|
371
|
+
chunk_kwargs = {
|
|
372
|
+
key: kwargs.pop(key)
|
|
373
|
+
for key in ("chunk", "chunk_profile", "chunk_strategy", "chunk_options")
|
|
374
|
+
if key in kwargs
|
|
375
|
+
}
|
|
376
|
+
documents = []
|
|
377
|
+
active_engine = engine or self._create_engine(engine_name, **kwargs)
|
|
378
|
+
for file_path in self.expand_inputs(inputs):
|
|
379
|
+
documents.append(
|
|
380
|
+
self.parse_file(
|
|
381
|
+
file_path,
|
|
382
|
+
engine_name=engine_name,
|
|
383
|
+
engine=active_engine,
|
|
384
|
+
workbook_disambiguation=workbook_disambiguation,
|
|
385
|
+
**(
|
|
386
|
+
{"progress_callback": progress_callback}
|
|
387
|
+
if progress_callback is not None
|
|
388
|
+
else {}
|
|
389
|
+
),
|
|
390
|
+
model=model,
|
|
391
|
+
api_key=api_key,
|
|
392
|
+
base_url=base_url,
|
|
393
|
+
**chunk_kwargs,
|
|
394
|
+
**kwargs,
|
|
395
|
+
)
|
|
396
|
+
)
|
|
397
|
+
return documents
|
|
398
|
+
|
|
399
|
+
def _collect_pdf_document_result(
|
|
400
|
+
self, file_path, engine_name="simple", engine=None, progress_callback=None, **kwargs
|
|
401
|
+
):
|
|
402
|
+
file_path = Path(file_path)
|
|
403
|
+
if not file_path.exists():
|
|
404
|
+
raise FileNotFoundError(f"File not found: {file_path}")
|
|
405
|
+
|
|
406
|
+
engine_kwargs = {
|
|
407
|
+
k: v
|
|
408
|
+
for k, v in kwargs.items()
|
|
409
|
+
if k
|
|
410
|
+
not in (
|
|
411
|
+
"model",
|
|
412
|
+
"api_key",
|
|
413
|
+
"base_url",
|
|
414
|
+
"workbook_disambiguation",
|
|
415
|
+
"disambiguation",
|
|
416
|
+
)
|
|
417
|
+
}
|
|
418
|
+
active_engine = engine or self._create_engine(engine_name, **engine_kwargs)
|
|
419
|
+
if progress_callback is not None:
|
|
420
|
+
kwargs["progress_callback"] = progress_callback
|
|
421
|
+
if hasattr(active_engine, "process_document"):
|
|
422
|
+
process_document = active_engine.process_document
|
|
423
|
+
if not callable(process_document):
|
|
424
|
+
raise TypeError(
|
|
425
|
+
f"{type(active_engine).__name__}.process_document exists but is not callable"
|
|
426
|
+
)
|
|
427
|
+
|
|
428
|
+
parsed = process_document(file_path, **kwargs)
|
|
429
|
+
if not isinstance(parsed, ParsedDocumentResult):
|
|
430
|
+
raise TypeError(
|
|
431
|
+
f"{type(active_engine).__name__}.process_document must return ParsedDocumentResult"
|
|
432
|
+
)
|
|
433
|
+
return parsed
|
|
434
|
+
|
|
435
|
+
pages = []
|
|
436
|
+
for page in active_engine.process(file_path, **kwargs):
|
|
437
|
+
pages.append(self._to_parsed_page_result(page))
|
|
438
|
+
|
|
439
|
+
return ParsedDocumentResult(
|
|
440
|
+
source=str(file_path),
|
|
441
|
+
filename=file_path.name,
|
|
442
|
+
engine=engine_name,
|
|
443
|
+
pages=pages,
|
|
444
|
+
markdown_content="\n".join(page.markdown_content for page in pages),
|
|
445
|
+
metadata=self._document_metadata_from_pages(pages),
|
|
446
|
+
)
|
|
447
|
+
|
|
448
|
+
def _document_metadata_from_pages(self, pages: list[ParsedPageResult]) -> dict:
|
|
449
|
+
"""Roll per-page engine signals up to the document, where metrics read them."""
|
|
450
|
+
return {
|
|
451
|
+
"ocr_applied": any(page.metadata.get("ocr_applied") for page in pages),
|
|
452
|
+
"ocr_text_chars": sum(
|
|
453
|
+
int(page.metadata.get("ocr_text_chars", 0) or 0) for page in pages
|
|
454
|
+
),
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
def create_engine(self, engine_name: str = "simple", **kwargs):
|
|
458
|
+
"""
|
|
459
|
+
Build one engine instance callers can reuse across many files.
|
|
460
|
+
|
|
461
|
+
Batch runs must share a single engine: a per-file MinerU engine would
|
|
462
|
+
start and stop its own local mineru-api service, and concurrent workers
|
|
463
|
+
would race for the same port.
|
|
464
|
+
"""
|
|
465
|
+
return self._create_engine(engine_name, **kwargs)
|
|
466
|
+
|
|
467
|
+
def _create_engine(self, engine_name: str, **kwargs):
|
|
468
|
+
engine_class = ENGINE_MAP.get(engine_name)
|
|
469
|
+
if engine_class is None:
|
|
470
|
+
available = ", ".join(sorted(ENGINE_MAP))
|
|
471
|
+
if engine_name in PLANNED_ENGINES:
|
|
472
|
+
raise ValueError(
|
|
473
|
+
f"Engine '{engine_name}' is not implemented yet. Available: {available}"
|
|
474
|
+
)
|
|
475
|
+
raise ValueError(f"Unknown engine: {engine_name}. Available: {available}")
|
|
476
|
+
|
|
477
|
+
engine_config = settings.resolve_engine_config(engine_name, kwargs)
|
|
478
|
+
return engine_class(**engine_config)
|
|
479
|
+
|
|
480
|
+
def _to_parsed_page_result(self, page) -> ParsedPageResult:
|
|
481
|
+
return ParsedPageResult(
|
|
482
|
+
page_number=page.page_number,
|
|
483
|
+
markdown_content=page.markdown_content,
|
|
484
|
+
plain_text=getattr(page, "plain_text", ""),
|
|
485
|
+
elements=list(getattr(page, "elements", [])),
|
|
486
|
+
tables=list(getattr(page, "tables", [])),
|
|
487
|
+
images=list(getattr(page, "images", [])),
|
|
488
|
+
metadata=dict(getattr(page, "metadata", {})),
|
|
489
|
+
)
|
|
490
|
+
|
|
491
|
+
def _build_document_from_result(self, parsed: ParsedDocumentResult) -> Document:
|
|
492
|
+
return document_from_result(parsed)
|
|
493
|
+
|
|
494
|
+
def _flatten_inputs(self, inputs) -> Iterator[str | Path]:
|
|
495
|
+
if isinstance(inputs, (str, Path)):
|
|
496
|
+
yield inputs
|
|
497
|
+
return
|
|
498
|
+
|
|
499
|
+
if isinstance(inputs, Iterable):
|
|
500
|
+
for item in inputs:
|
|
501
|
+
if isinstance(item, (str, Path)):
|
|
502
|
+
yield item
|
|
503
|
+
elif isinstance(item, Iterable):
|
|
504
|
+
yield from self._flatten_inputs(item)
|
|
505
|
+
else:
|
|
506
|
+
yield item
|
|
507
|
+
return
|
|
508
|
+
|
|
509
|
+
yield inputs
|
|
510
|
+
|
|
511
|
+
def _output_filename(self, source, fmt: str) -> str:
|
|
512
|
+
return output_filename(source, fmt)
|
|
513
|
+
|
|
514
|
+
def _output_path_for_batch_item(self, source, fmt: str, used_paths: set[Path]) -> Path:
|
|
515
|
+
return resolve_output_path(source, fmt, used_paths)
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
def _json_scalar(value):
|
|
519
|
+
"""Serialize native spreadsheet scalars such as dates and decimals."""
|
|
520
|
+
|
|
521
|
+
if hasattr(value, "isoformat"):
|
|
522
|
+
return value.isoformat()
|
|
523
|
+
return str(value)
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
|
|
5
|
+
from langparse.metrics import ParseMetrics
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass
|
|
9
|
+
class QualityCheck:
|
|
10
|
+
min_pages: int | None = None
|
|
11
|
+
min_chars: int | None = None
|
|
12
|
+
min_tables: int | None = None
|
|
13
|
+
min_images: int | None = None
|
|
14
|
+
require_page_markers: bool = False
|
|
15
|
+
require_table_markdown: bool = False
|
|
16
|
+
require_ocr_text: bool = False
|
|
17
|
+
require_multi_column_check: bool = False
|
|
18
|
+
max_header_footer_repetition_ratio: float | None = None
|
|
19
|
+
require_captions_for_images: bool = False
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class QualityCheckResult:
|
|
24
|
+
passed: bool
|
|
25
|
+
failures: list[str] = field(default_factory=list)
|
|
26
|
+
warnings: list[str] = field(default_factory=list)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def run_quality_checks(metrics: ParseMetrics, checks: QualityCheck) -> QualityCheckResult:
|
|
30
|
+
failures: list[str] = []
|
|
31
|
+
warnings: list[str] = []
|
|
32
|
+
|
|
33
|
+
if checks.min_pages is not None and metrics.page_count < checks.min_pages:
|
|
34
|
+
failures.append("min_pages")
|
|
35
|
+
if checks.min_chars is not None and metrics.markdown_chars < checks.min_chars:
|
|
36
|
+
failures.append("min_chars")
|
|
37
|
+
if checks.min_tables is not None and metrics.table_count < checks.min_tables:
|
|
38
|
+
failures.append("min_tables")
|
|
39
|
+
if checks.min_images is not None and metrics.image_count < checks.min_images:
|
|
40
|
+
failures.append("min_images")
|
|
41
|
+
if checks.require_page_markers and metrics.page_marker_coverage <= 0:
|
|
42
|
+
failures.append("require_page_markers")
|
|
43
|
+
if checks.require_table_markdown and metrics.table_count <= 0:
|
|
44
|
+
failures.append("require_table_markdown")
|
|
45
|
+
if checks.require_ocr_text and metrics.ocr_text_chars <= 0:
|
|
46
|
+
failures.append("require_ocr_text")
|
|
47
|
+
if (
|
|
48
|
+
checks.require_multi_column_check
|
|
49
|
+
and not metrics.multi_column_detected
|
|
50
|
+
and metrics.reading_order_warnings == 0
|
|
51
|
+
):
|
|
52
|
+
failures.append("require_multi_column_check")
|
|
53
|
+
if (
|
|
54
|
+
checks.require_captions_for_images
|
|
55
|
+
and metrics.image_count > 0
|
|
56
|
+
and metrics.images_with_caption_ratio < 1.0
|
|
57
|
+
):
|
|
58
|
+
failures.append("require_captions_for_images")
|
|
59
|
+
if (
|
|
60
|
+
checks.max_header_footer_repetition_ratio is not None
|
|
61
|
+
and metrics.header_footer_removed_count == 0
|
|
62
|
+
):
|
|
63
|
+
warnings.append("header_footer_filter_not_applied")
|
|
64
|
+
|
|
65
|
+
return QualityCheckResult(passed=not failures, failures=failures, warnings=warnings)
|