langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
import threading
|
|
2
|
+
from collections.abc import Iterator
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
from langparse.core.engine import PageResult
|
|
7
|
+
from langparse.engines.pdf.simple import BasePDFEngine
|
|
8
|
+
from langparse.logging import get_logger
|
|
9
|
+
from langparse.progress import ProgressCallback, ProgressReporter
|
|
10
|
+
from langparse.types import ParsedDocumentResult
|
|
11
|
+
|
|
12
|
+
logger = get_logger(__name__)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class DeepDocEngine(BasePDFEngine):
|
|
16
|
+
"""
|
|
17
|
+
Adapter for the ported DeepDoc (RAGFlow) OCR + layout + table-structure
|
|
18
|
+
pipeline. CPU-only ONNX inference, no separate runtime service --
|
|
19
|
+
unlike MinerU, it runs in-process (see langparse/engines/pdf/deepdoc/).
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
def __init__(
|
|
23
|
+
self,
|
|
24
|
+
device: str = "cpu",
|
|
25
|
+
model_dir: str | None = None,
|
|
26
|
+
download_dir: str | None = None,
|
|
27
|
+
model_policy: str = "download_if_missing",
|
|
28
|
+
parser: Any = None,
|
|
29
|
+
**kwargs: Any,
|
|
30
|
+
):
|
|
31
|
+
if device != "cpu":
|
|
32
|
+
raise ValueError(
|
|
33
|
+
f"DeepDocEngine only supports device='cpu' in this version, got: {device!r}"
|
|
34
|
+
)
|
|
35
|
+
self.device = device
|
|
36
|
+
self.model_dir = model_dir
|
|
37
|
+
self.download_dir = download_dir
|
|
38
|
+
self.model_policy = model_policy
|
|
39
|
+
self._parser = parser
|
|
40
|
+
# Batch runs share one engine (and thus one RAGFlowPdfParser) across
|
|
41
|
+
# worker threads. RAGFlowPdfParser is deeply per-parse stateful --
|
|
42
|
+
# __images__ resets self.boxes/self.page_images/etc. at the top of
|
|
43
|
+
# every call, and the downstream chain mutates self.boxes in place --
|
|
44
|
+
# so two threads parsing concurrently would corrupt each other's
|
|
45
|
+
# state. The lock also prevents two threads from racing to build (and
|
|
46
|
+
# load ~100MB of models for) separate parsers. Same precedent as
|
|
47
|
+
# SimplePDFEngine._ocr_lock: correctness over throughput on an
|
|
48
|
+
# already-slow path.
|
|
49
|
+
self._parser_lock = threading.Lock()
|
|
50
|
+
|
|
51
|
+
def _build_parser(self):
|
|
52
|
+
try:
|
|
53
|
+
from langparse.engines.pdf.deepdoc.model_loader import ensure_deepdoc_models
|
|
54
|
+
from langparse.engines.pdf.deepdoc.pdf_parser import RAGFlowPdfParser
|
|
55
|
+
except ImportError as exc:
|
|
56
|
+
raise ImportError(
|
|
57
|
+
'DeepDoc engine needs extra dependencies. Install them with `pip install "langparse[deepdoc]"`.'
|
|
58
|
+
) from exc
|
|
59
|
+
|
|
60
|
+
resolved_model_dir = ensure_deepdoc_models(
|
|
61
|
+
model_dir=self.model_dir,
|
|
62
|
+
download_dir=self.download_dir,
|
|
63
|
+
model_policy=self.model_policy,
|
|
64
|
+
)
|
|
65
|
+
return RAGFlowPdfParser(model_dir=resolved_model_dir)
|
|
66
|
+
|
|
67
|
+
def _classify_ocr_pages(self, file_path: Path) -> dict[int, bool]:
|
|
68
|
+
"""Per-page {page_number: bool} saying whether a page's text should be
|
|
69
|
+
credited to OCR, using the same needs_ocr() heuristic simple/ocr.py
|
|
70
|
+
already uses -- independent of deepdoc's own (unconditional) internal
|
|
71
|
+
OCR pass, so the two engines report comparable metadata for comparable
|
|
72
|
+
input. 1-indexed to match render_pages()'s page_number convention;
|
|
73
|
+
pdfplumber.pages itself is 0-indexed.
|
|
74
|
+
|
|
75
|
+
This runs after the real (and potentially minutes-long) deepdoc parse
|
|
76
|
+
has already completed, purely to annotate advisory metadata. Any
|
|
77
|
+
failure here (missing pdfplumber, malformed/encrypted/oddly-structured
|
|
78
|
+
PDF, etc.) is logged and swallowed rather than raised, so it can never
|
|
79
|
+
discard an otherwise-successful parse. An empty dict is already the
|
|
80
|
+
documented contract: render_pages() defaults every absent page to
|
|
81
|
+
ocr_applied=False / ocr_text_chars=0.
|
|
82
|
+
"""
|
|
83
|
+
try:
|
|
84
|
+
try:
|
|
85
|
+
import pdfplumber
|
|
86
|
+
except ImportError as exc:
|
|
87
|
+
raise ImportError(
|
|
88
|
+
'DeepDoc engine needs extra dependencies. Install them with `pip install "langparse[deepdoc]"`.'
|
|
89
|
+
) from exc
|
|
90
|
+
from langparse.engines.pdf.ocr import needs_ocr
|
|
91
|
+
|
|
92
|
+
with pdfplumber.open(file_path) as pdf:
|
|
93
|
+
return {
|
|
94
|
+
page_number: needs_ocr(page)
|
|
95
|
+
for page_number, page in enumerate(pdf.pages, start=1)
|
|
96
|
+
}
|
|
97
|
+
except Exception as exc:
|
|
98
|
+
logger.warning("Skipping OCR-applied classification for %s: %s", file_path, exc)
|
|
99
|
+
return {}
|
|
100
|
+
|
|
101
|
+
def process_document(
|
|
102
|
+
self,
|
|
103
|
+
file_path: Path,
|
|
104
|
+
*,
|
|
105
|
+
progress_callback: ProgressCallback | None = None,
|
|
106
|
+
**kwargs: Any,
|
|
107
|
+
) -> ParsedDocumentResult:
|
|
108
|
+
reporter = ProgressReporter(str(file_path), progress_callback)
|
|
109
|
+
reporter.emit("preparing", message="Preparing DeepDoc parser")
|
|
110
|
+
try:
|
|
111
|
+
from langparse.engines.pdf.deepdoc.rendering import render_pages
|
|
112
|
+
except ImportError as exc:
|
|
113
|
+
raise ImportError(
|
|
114
|
+
'DeepDoc engine needs extra dependencies. Install them with `pip install "langparse[deepdoc]"`.'
|
|
115
|
+
) from exc
|
|
116
|
+
|
|
117
|
+
with self._parser_lock:
|
|
118
|
+
if self._parser is None:
|
|
119
|
+
self._parser = self._build_parser()
|
|
120
|
+
if progress_callback is None:
|
|
121
|
+
boxes = self._parser.parse_into_bboxes(str(file_path))
|
|
122
|
+
else:
|
|
123
|
+
last_percent = 0.0
|
|
124
|
+
|
|
125
|
+
def on_progress(progress, message=""):
|
|
126
|
+
nonlocal last_percent
|
|
127
|
+
last_percent = max(last_percent, min(100.0, max(0.0, progress * 100)))
|
|
128
|
+
reporter.emit("parsing", percent=last_percent, message=message)
|
|
129
|
+
|
|
130
|
+
boxes = self._parser.parse_into_bboxes(str(file_path), callback=on_progress)
|
|
131
|
+
|
|
132
|
+
reporter.emit("rendering")
|
|
133
|
+
ocr_pages = self._classify_ocr_pages(file_path)
|
|
134
|
+
pages = render_pages(boxes, ocr_pages=ocr_pages)
|
|
135
|
+
return ParsedDocumentResult(
|
|
136
|
+
source=str(file_path),
|
|
137
|
+
filename=Path(file_path).name,
|
|
138
|
+
engine="deepdoc",
|
|
139
|
+
pages=pages,
|
|
140
|
+
markdown_content="\n\n".join(page.markdown_content for page in pages),
|
|
141
|
+
metadata={
|
|
142
|
+
"device": self.device,
|
|
143
|
+
"model_dir": self.model_dir,
|
|
144
|
+
"ocr_applied": any(page.metadata["ocr_applied"] for page in pages),
|
|
145
|
+
"ocr_text_chars": sum(page.metadata["ocr_text_chars"] for page in pages),
|
|
146
|
+
},
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
def process(
|
|
150
|
+
self, file_path: Path, *, progress_callback: ProgressCallback | None = None, **kwargs
|
|
151
|
+
) -> Iterator[PageResult]:
|
|
152
|
+
if progress_callback is not None:
|
|
153
|
+
kwargs["progress_callback"] = progress_callback
|
|
154
|
+
parsed = self.process_document(file_path, **kwargs)
|
|
155
|
+
for page in parsed.pages:
|
|
156
|
+
yield PageResult(
|
|
157
|
+
page_number=page.page_number,
|
|
158
|
+
markdown_content=page.markdown_content,
|
|
159
|
+
plain_text=page.plain_text,
|
|
160
|
+
elements=page.elements,
|
|
161
|
+
tables=page.tables,
|
|
162
|
+
images=page.images,
|
|
163
|
+
metadata=page.metadata,
|
|
164
|
+
)
|
|
@@ -0,0 +1,259 @@
|
|
|
1
|
+
from collections.abc import Iterator, Mapping
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
from langparse.core.engine import PageResult
|
|
6
|
+
from langparse.engines.pdf.mineru_client import MinerUClient
|
|
7
|
+
from langparse.engines.pdf.mineru_service import MinerUServiceManager
|
|
8
|
+
from langparse.engines.pdf.simple import BasePDFEngine
|
|
9
|
+
from langparse.progress import ProgressCallback, ProgressReporter
|
|
10
|
+
from langparse.types import ParsedDocumentResult, ParsedElement, ParsedPageResult
|
|
11
|
+
|
|
12
|
+
_SENSITIVE_OPTION_KEYS = frozenset(
|
|
13
|
+
{
|
|
14
|
+
"api_key",
|
|
15
|
+
"openai_api_key",
|
|
16
|
+
"model",
|
|
17
|
+
"base_url",
|
|
18
|
+
"openai_base_url",
|
|
19
|
+
"openai_model",
|
|
20
|
+
"workbook_disambiguation",
|
|
21
|
+
"disambiguation",
|
|
22
|
+
"token",
|
|
23
|
+
"secret",
|
|
24
|
+
"apikey",
|
|
25
|
+
"auth",
|
|
26
|
+
"authorization",
|
|
27
|
+
"password",
|
|
28
|
+
"bearer",
|
|
29
|
+
"credentials",
|
|
30
|
+
}
|
|
31
|
+
)
|
|
32
|
+
_SENSITIVE_OPTION_SUFFIXES = ("_key", "_secret", "_token", "_password", "_auth")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _sanitize_extra_options(options: Mapping[str, Any] | None) -> dict[str, Any]:
|
|
36
|
+
if options is None:
|
|
37
|
+
return {}
|
|
38
|
+
return {
|
|
39
|
+
key: value
|
|
40
|
+
for key, value in options.items()
|
|
41
|
+
if key.lower() not in _SENSITIVE_OPTION_KEYS
|
|
42
|
+
and not key.lower().endswith(_SENSITIVE_OPTION_SUFFIXES)
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class MinerUEngine(BasePDFEngine):
|
|
47
|
+
"""
|
|
48
|
+
Adapter for MinerU (Magic-PDF).
|
|
49
|
+
High precision parsing for complex documents (papers, textbooks).
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
def __init__(
|
|
53
|
+
self,
|
|
54
|
+
device: str = "auto",
|
|
55
|
+
model_dir: str | None = None,
|
|
56
|
+
download_dir: str | None = None,
|
|
57
|
+
enable_ocr: bool = True,
|
|
58
|
+
api_url: str | None = None,
|
|
59
|
+
api_host: str = "127.0.0.1",
|
|
60
|
+
api_port: int = 8000,
|
|
61
|
+
api_command: str = "mineru-api",
|
|
62
|
+
api_start_timeout: float = 30.0,
|
|
63
|
+
request_timeout: float = 300.0,
|
|
64
|
+
backend: str | None = None,
|
|
65
|
+
server_url: str | None = None,
|
|
66
|
+
model_policy: str = "download_if_missing",
|
|
67
|
+
model_source: str | None = None,
|
|
68
|
+
auto_install_runtime: bool = False,
|
|
69
|
+
runtime_package: str = "mineru>=3.4,<4",
|
|
70
|
+
extra_options: dict[str, Any] | None = None,
|
|
71
|
+
**kwargs: Any,
|
|
72
|
+
):
|
|
73
|
+
self.device = device
|
|
74
|
+
self.model_dir = model_dir
|
|
75
|
+
self.download_dir = download_dir
|
|
76
|
+
self.enable_ocr = enable_ocr
|
|
77
|
+
self.api_url = api_url
|
|
78
|
+
self.api_host = api_host
|
|
79
|
+
self.api_port = api_port
|
|
80
|
+
self.api_command = api_command
|
|
81
|
+
self.api_start_timeout = api_start_timeout
|
|
82
|
+
self.request_timeout = request_timeout
|
|
83
|
+
self.backend = backend
|
|
84
|
+
self.server_url = server_url
|
|
85
|
+
self.model_policy = model_policy
|
|
86
|
+
self.model_source = model_source
|
|
87
|
+
self.auto_install_runtime = auto_install_runtime
|
|
88
|
+
self.runtime_package = runtime_package
|
|
89
|
+
self.extra_options = {
|
|
90
|
+
**_sanitize_extra_options(extra_options),
|
|
91
|
+
**_sanitize_extra_options(kwargs),
|
|
92
|
+
}
|
|
93
|
+
if self.backend is not None:
|
|
94
|
+
self.extra_options["backend"] = self.backend
|
|
95
|
+
if self.server_url is not None:
|
|
96
|
+
self.extra_options["server_url"] = self.server_url
|
|
97
|
+
|
|
98
|
+
def _cuda_available(self) -> bool:
|
|
99
|
+
try:
|
|
100
|
+
import torch
|
|
101
|
+
except ImportError:
|
|
102
|
+
return False
|
|
103
|
+
|
|
104
|
+
return bool(torch.cuda.is_available())
|
|
105
|
+
|
|
106
|
+
def _resolve_device(self, device: str | None = None) -> str:
|
|
107
|
+
requested_device = self.device if device is None else device
|
|
108
|
+
|
|
109
|
+
if requested_device == "auto":
|
|
110
|
+
return "cuda" if self._cuda_available() else "cpu"
|
|
111
|
+
|
|
112
|
+
if requested_device == "cuda" and not self._cuda_available():
|
|
113
|
+
raise RuntimeError("CUDA was requested for MinerU but is not available.")
|
|
114
|
+
|
|
115
|
+
return requested_device
|
|
116
|
+
|
|
117
|
+
def _ensure_runtime(self) -> None:
|
|
118
|
+
# The real runtime path is mineru-api. Dependency validation happens when
|
|
119
|
+
# starting or connecting to that service.
|
|
120
|
+
return None
|
|
121
|
+
|
|
122
|
+
def _build_runtime_config(self, **kwargs: Any) -> dict[str, Any]:
|
|
123
|
+
requested_device = kwargs.get("device", self.device)
|
|
124
|
+
runtime_extra_options = _sanitize_extra_options(kwargs.get("extra_options"))
|
|
125
|
+
return {
|
|
126
|
+
"device": self._resolve_device(requested_device),
|
|
127
|
+
"model_dir": kwargs.get("model_dir", self.model_dir),
|
|
128
|
+
"download_dir": kwargs.get("download_dir", self.download_dir),
|
|
129
|
+
"enable_ocr": kwargs.get("enable_ocr", self.enable_ocr),
|
|
130
|
+
"extra_options": {**self.extra_options, **runtime_extra_options},
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
def _build_service_config(self) -> dict[str, Any]:
|
|
134
|
+
return {
|
|
135
|
+
"api_url": self.api_url,
|
|
136
|
+
"host": self.api_host,
|
|
137
|
+
"port": self.api_port,
|
|
138
|
+
"command": self.api_command,
|
|
139
|
+
"start_timeout": self.api_start_timeout,
|
|
140
|
+
"request_timeout": self.request_timeout,
|
|
141
|
+
"model_dir": self.model_dir,
|
|
142
|
+
"download_dir": self.download_dir,
|
|
143
|
+
"model_policy": self.model_policy,
|
|
144
|
+
"model_source": self.model_source,
|
|
145
|
+
"auto_install_runtime": self.auto_install_runtime,
|
|
146
|
+
"runtime_package": self.runtime_package,
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
def _create_client(self, base_url: str) -> MinerUClient:
|
|
150
|
+
return MinerUClient(base_url, timeout=self.request_timeout)
|
|
151
|
+
|
|
152
|
+
def _create_service_manager(self) -> MinerUServiceManager:
|
|
153
|
+
return MinerUServiceManager(**self._build_service_config())
|
|
154
|
+
|
|
155
|
+
def _run_mineru(
|
|
156
|
+
self,
|
|
157
|
+
file_path: Path,
|
|
158
|
+
runtime_config: dict[str, Any],
|
|
159
|
+
progress_callback: ProgressCallback | None = None,
|
|
160
|
+
) -> list[dict[str, Any]]:
|
|
161
|
+
reporter = ProgressReporter(str(file_path), progress_callback)
|
|
162
|
+
manager = self._create_service_manager()
|
|
163
|
+
with manager.running_service() as base_url:
|
|
164
|
+
client = self._create_client(base_url)
|
|
165
|
+
reporter.emit("parsing", message="Waiting for MinerU response")
|
|
166
|
+
return client.parse_file(file_path, runtime_config)
|
|
167
|
+
|
|
168
|
+
def process_document(
|
|
169
|
+
self,
|
|
170
|
+
file_path: Path,
|
|
171
|
+
*,
|
|
172
|
+
progress_callback: ProgressCallback | None = None,
|
|
173
|
+
**kwargs: Any,
|
|
174
|
+
) -> ParsedDocumentResult:
|
|
175
|
+
reporter = ProgressReporter(str(file_path), progress_callback)
|
|
176
|
+
reporter.emit("preparing", message="Preparing MinerU service")
|
|
177
|
+
self._ensure_runtime()
|
|
178
|
+
runtime_config = self._build_runtime_config(**kwargs)
|
|
179
|
+
if progress_callback is None:
|
|
180
|
+
raw_pages = self._run_mineru(file_path, runtime_config)
|
|
181
|
+
else:
|
|
182
|
+
raw_pages = self._run_mineru(file_path, runtime_config, progress_callback)
|
|
183
|
+
reporter.emit("rendering")
|
|
184
|
+
pages = [
|
|
185
|
+
ParsedPageResult(
|
|
186
|
+
page_number=item["page_number"],
|
|
187
|
+
markdown_content=item.get("markdown", ""),
|
|
188
|
+
plain_text=item.get("plain_text", ""),
|
|
189
|
+
elements=[
|
|
190
|
+
element
|
|
191
|
+
if isinstance(element, ParsedElement)
|
|
192
|
+
else ParsedElement(
|
|
193
|
+
kind=element.get("kind", "text"),
|
|
194
|
+
text=element.get("text", ""),
|
|
195
|
+
bbox=element.get("bbox"),
|
|
196
|
+
metadata=element.get("metadata", {}),
|
|
197
|
+
)
|
|
198
|
+
for element in item.get("elements", [])
|
|
199
|
+
],
|
|
200
|
+
tables=item.get("tables", []),
|
|
201
|
+
images=item.get("images", []),
|
|
202
|
+
metadata={
|
|
203
|
+
"engine_name": "mineru",
|
|
204
|
+
"device": runtime_config["device"],
|
|
205
|
+
"engine_specific": item.get("engine_specific", {}),
|
|
206
|
+
},
|
|
207
|
+
)
|
|
208
|
+
for item in raw_pages
|
|
209
|
+
]
|
|
210
|
+
engine_specific_items = [page.metadata.get("engine_specific", {}) for page in pages]
|
|
211
|
+
quality_metadata = {
|
|
212
|
+
"ocr_applied": any(bool(item.get("ocr_applied")) for item in engine_specific_items),
|
|
213
|
+
"ocr_text_chars": sum(
|
|
214
|
+
int(item.get("ocr_text_chars", 0) or 0) for item in engine_specific_items
|
|
215
|
+
),
|
|
216
|
+
"multi_column_detected": any(
|
|
217
|
+
bool(item.get("multi_column_detected")) for item in engine_specific_items
|
|
218
|
+
),
|
|
219
|
+
"reading_order_warnings": sum(
|
|
220
|
+
int(item.get("reading_order_warnings", 0) or 0) for item in engine_specific_items
|
|
221
|
+
),
|
|
222
|
+
"header_footer_removed_count": sum(
|
|
223
|
+
int(item.get("header_footer_removed_count", 0) or 0)
|
|
224
|
+
for item in engine_specific_items
|
|
225
|
+
),
|
|
226
|
+
}
|
|
227
|
+
return ParsedDocumentResult(
|
|
228
|
+
source=str(file_path),
|
|
229
|
+
filename=file_path.name,
|
|
230
|
+
engine="mineru",
|
|
231
|
+
pages=pages,
|
|
232
|
+
markdown_content="\n".join(page.markdown_content for page in pages),
|
|
233
|
+
metadata={
|
|
234
|
+
"device": runtime_config["device"],
|
|
235
|
+
"model_dir": runtime_config["model_dir"],
|
|
236
|
+
"download_dir": runtime_config["download_dir"],
|
|
237
|
+
"enable_ocr": runtime_config["enable_ocr"],
|
|
238
|
+
"model_policy": self.model_policy,
|
|
239
|
+
"model_source": self.model_source,
|
|
240
|
+
**quality_metadata,
|
|
241
|
+
},
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
def process(
|
|
245
|
+
self, file_path: Path, *, progress_callback: ProgressCallback | None = None, **kwargs
|
|
246
|
+
) -> Iterator[PageResult]:
|
|
247
|
+
if progress_callback is not None:
|
|
248
|
+
kwargs["progress_callback"] = progress_callback
|
|
249
|
+
parsed = self.process_document(file_path, **kwargs)
|
|
250
|
+
for page in parsed.pages:
|
|
251
|
+
yield PageResult(
|
|
252
|
+
page_number=page.page_number,
|
|
253
|
+
markdown_content=page.markdown_content,
|
|
254
|
+
plain_text=page.plain_text,
|
|
255
|
+
elements=page.elements,
|
|
256
|
+
tables=page.tables,
|
|
257
|
+
images=page.images,
|
|
258
|
+
metadata=page.metadata,
|
|
259
|
+
)
|