langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,318 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import mimetypes
|
|
5
|
+
import uuid
|
|
6
|
+
from html.parser import HTMLParser
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any
|
|
9
|
+
from urllib import request
|
|
10
|
+
from urllib.error import HTTPError, URLError
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class _HTMLTableReader(HTMLParser):
|
|
14
|
+
"""Collects rows/cells from the HTML fragment MinerU returns as ``table_body``."""
|
|
15
|
+
|
|
16
|
+
def __init__(self):
|
|
17
|
+
super().__init__(convert_charrefs=True)
|
|
18
|
+
self.rows: list[list[str]] = []
|
|
19
|
+
self._row: list[str] | None = None
|
|
20
|
+
self._cell: list[str] | None = None
|
|
21
|
+
|
|
22
|
+
def handle_starttag(self, tag, attrs):
|
|
23
|
+
if tag == "tr":
|
|
24
|
+
self._row = []
|
|
25
|
+
elif tag in ("td", "th"):
|
|
26
|
+
self._cell = []
|
|
27
|
+
|
|
28
|
+
def handle_endtag(self, tag):
|
|
29
|
+
if tag in ("td", "th") and self._cell is not None:
|
|
30
|
+
text = " ".join("".join(self._cell).split())
|
|
31
|
+
if self._row is None:
|
|
32
|
+
self._row = []
|
|
33
|
+
self._row.append(text)
|
|
34
|
+
self._cell = None
|
|
35
|
+
elif tag == "tr" and self._row is not None:
|
|
36
|
+
self.rows.append(self._row)
|
|
37
|
+
self._row = None
|
|
38
|
+
|
|
39
|
+
def handle_data(self, data):
|
|
40
|
+
if self._cell is not None:
|
|
41
|
+
self._cell.append(data)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def parse_html_table(markup: str) -> list[list[str]]:
|
|
45
|
+
"""Parse an HTML table fragment into rows of cell text. Returns [] if unparseable."""
|
|
46
|
+
if not markup:
|
|
47
|
+
return []
|
|
48
|
+
reader = _HTMLTableReader()
|
|
49
|
+
try:
|
|
50
|
+
reader.feed(markup)
|
|
51
|
+
reader.close()
|
|
52
|
+
except Exception:
|
|
53
|
+
return []
|
|
54
|
+
return [row for row in reader.rows if row]
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def rows_to_markdown(rows: list[list[str]]) -> str:
|
|
58
|
+
"""Render rows as a Markdown table, padding short rows to the header width."""
|
|
59
|
+
if not rows:
|
|
60
|
+
return ""
|
|
61
|
+
width = max(len(row) for row in rows)
|
|
62
|
+
padded = [row + [""] * (width - len(row)) for row in rows]
|
|
63
|
+
lines = [f"| {' | '.join(padded[0])} |", f"| {' | '.join(['---'] * width)} |"]
|
|
64
|
+
lines.extend(f"| {' | '.join(row)} |" for row in padded[1:])
|
|
65
|
+
return "\n".join(lines)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class MinerUClient:
|
|
69
|
+
def __init__(self, base_url: str, timeout: float = 300.0):
|
|
70
|
+
self.base_url = base_url.rstrip("/")
|
|
71
|
+
self.timeout = timeout
|
|
72
|
+
|
|
73
|
+
def health(self) -> dict[str, Any]:
|
|
74
|
+
return self._request_json("GET", "/health")
|
|
75
|
+
|
|
76
|
+
def parse_file(self, file_path: Path, runtime_config: dict[str, Any]) -> list[dict[str, Any]]:
|
|
77
|
+
response = self._request_json(
|
|
78
|
+
"POST",
|
|
79
|
+
"/file_parse",
|
|
80
|
+
fields=self._build_form_fields(runtime_config),
|
|
81
|
+
file_path=file_path,
|
|
82
|
+
)
|
|
83
|
+
return self._normalize_parse_response(response)
|
|
84
|
+
|
|
85
|
+
def _build_form_fields(self, runtime_config: dict[str, Any]) -> dict[str, str]:
|
|
86
|
+
fields = {
|
|
87
|
+
"return_md": "true",
|
|
88
|
+
"return_content_list": "true",
|
|
89
|
+
"response_format_zip": "false",
|
|
90
|
+
}
|
|
91
|
+
extra_options = runtime_config.get("extra_options", {})
|
|
92
|
+
if runtime_config.get("enable_ocr") is False:
|
|
93
|
+
fields["parse_method"] = "txt"
|
|
94
|
+
if runtime_config.get("device"):
|
|
95
|
+
fields["device"] = str(runtime_config["device"])
|
|
96
|
+
if runtime_config.get("model_dir"):
|
|
97
|
+
fields["model_dir"] = str(runtime_config["model_dir"])
|
|
98
|
+
if runtime_config.get("download_dir"):
|
|
99
|
+
fields["download_dir"] = str(runtime_config["download_dir"])
|
|
100
|
+
for key, value in extra_options.items():
|
|
101
|
+
if value is None:
|
|
102
|
+
continue
|
|
103
|
+
fields[str(key)] = str(value)
|
|
104
|
+
return fields
|
|
105
|
+
|
|
106
|
+
def _request_json(
|
|
107
|
+
self,
|
|
108
|
+
method: str,
|
|
109
|
+
path: str,
|
|
110
|
+
fields: dict[str, str] | None = None,
|
|
111
|
+
file_path: Path | None = None,
|
|
112
|
+
) -> dict[str, Any]:
|
|
113
|
+
headers = {"Accept": "application/json"}
|
|
114
|
+
data = None
|
|
115
|
+
if file_path is not None:
|
|
116
|
+
data, content_type = self._encode_multipart_form(fields or {}, file_path)
|
|
117
|
+
headers["Content-Type"] = content_type
|
|
118
|
+
|
|
119
|
+
req = request.Request(
|
|
120
|
+
f"{self.base_url}{path}",
|
|
121
|
+
data=data,
|
|
122
|
+
headers=headers,
|
|
123
|
+
method=method,
|
|
124
|
+
)
|
|
125
|
+
try:
|
|
126
|
+
with request.urlopen(req, timeout=self.timeout) as response:
|
|
127
|
+
payload = response.read().decode("utf-8")
|
|
128
|
+
except HTTPError as exc:
|
|
129
|
+
detail = exc.read().decode("utf-8", errors="replace")
|
|
130
|
+
raise RuntimeError(f"MinerU API request failed with HTTP {exc.code}: {detail}") from exc
|
|
131
|
+
except URLError as exc:
|
|
132
|
+
raise RuntimeError(f"MinerU API request failed: {exc.reason}") from exc
|
|
133
|
+
|
|
134
|
+
try:
|
|
135
|
+
return json.loads(payload)
|
|
136
|
+
except json.JSONDecodeError as exc:
|
|
137
|
+
raise RuntimeError("MinerU API returned a non-JSON response.") from exc
|
|
138
|
+
|
|
139
|
+
def _encode_multipart_form(self, fields: dict[str, str], file_path: Path) -> tuple[bytes, str]:
|
|
140
|
+
boundary = f"----langparse-mineru-{uuid.uuid4().hex}"
|
|
141
|
+
lines: list[bytes] = []
|
|
142
|
+
for name, value in fields.items():
|
|
143
|
+
lines.extend(
|
|
144
|
+
[
|
|
145
|
+
f"--{boundary}".encode(),
|
|
146
|
+
f'Content-Disposition: form-data; name="{name}"'.encode(),
|
|
147
|
+
b"",
|
|
148
|
+
str(value).encode("utf-8"),
|
|
149
|
+
]
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
content_type = mimetypes.guess_type(file_path.name)[0] or "application/octet-stream"
|
|
153
|
+
file_bytes = file_path.read_bytes()
|
|
154
|
+
lines.extend(
|
|
155
|
+
[
|
|
156
|
+
f"--{boundary}".encode(),
|
|
157
|
+
f'Content-Disposition: form-data; name="files"; filename="{file_path.name}"'.encode(),
|
|
158
|
+
f"Content-Type: {content_type}".encode(),
|
|
159
|
+
b"",
|
|
160
|
+
file_bytes,
|
|
161
|
+
f"--{boundary}--".encode(),
|
|
162
|
+
b"",
|
|
163
|
+
]
|
|
164
|
+
)
|
|
165
|
+
return b"\r\n".join(lines), f"multipart/form-data; boundary={boundary}"
|
|
166
|
+
|
|
167
|
+
def _normalize_parse_response(self, response: dict[str, Any]) -> list[dict[str, Any]]:
|
|
168
|
+
markdown = self._extract_markdown(response)
|
|
169
|
+
content_list = self._extract_content_list(response)
|
|
170
|
+
if not content_list:
|
|
171
|
+
return [{"page_number": 1, "markdown": markdown}]
|
|
172
|
+
|
|
173
|
+
page_map: dict[int, list[dict[str, Any]]] = {}
|
|
174
|
+
for item in content_list:
|
|
175
|
+
page_idx = int(item.get("page_idx", 0))
|
|
176
|
+
page_map.setdefault(page_idx, []).append(item)
|
|
177
|
+
|
|
178
|
+
pages = []
|
|
179
|
+
for page_idx in sorted(page_map):
|
|
180
|
+
items = page_map[page_idx]
|
|
181
|
+
pages.append(self._build_page(page_idx, items, markdown))
|
|
182
|
+
return pages
|
|
183
|
+
|
|
184
|
+
def _build_page(
|
|
185
|
+
self, page_idx: int, items: list[dict[str, Any]], document_markdown: str
|
|
186
|
+
) -> dict[str, Any]:
|
|
187
|
+
markdown_blocks: list[str] = []
|
|
188
|
+
text_lines: list[str] = []
|
|
189
|
+
tables: list[dict[str, Any]] = []
|
|
190
|
+
images: list[dict[str, Any]] = []
|
|
191
|
+
elements: list[dict[str, Any]] = []
|
|
192
|
+
|
|
193
|
+
for item in items:
|
|
194
|
+
kind = item.get("type", "text")
|
|
195
|
+
caption = self._join_caption(item.get(f"{kind}_caption"))
|
|
196
|
+
|
|
197
|
+
if kind == "table":
|
|
198
|
+
rows = parse_html_table(item.get("table_body", ""))
|
|
199
|
+
table_markdown = rows_to_markdown(rows)
|
|
200
|
+
tables.append(
|
|
201
|
+
{
|
|
202
|
+
"rows": rows,
|
|
203
|
+
"caption": caption,
|
|
204
|
+
"html": item.get("table_body", ""),
|
|
205
|
+
"img_path": item.get("img_path"),
|
|
206
|
+
}
|
|
207
|
+
)
|
|
208
|
+
block = "\n\n".join(part for part in (caption, table_markdown) if part)
|
|
209
|
+
if block:
|
|
210
|
+
markdown_blocks.append(block)
|
|
211
|
+
element_text = table_markdown
|
|
212
|
+
elif kind == "image":
|
|
213
|
+
images.append(
|
|
214
|
+
{
|
|
215
|
+
"path": item.get("img_path"),
|
|
216
|
+
"caption": caption,
|
|
217
|
+
"footnote": self._join_caption(item.get("image_footnote")),
|
|
218
|
+
}
|
|
219
|
+
)
|
|
220
|
+
block = f" or ''})"
|
|
221
|
+
markdown_blocks.append(block)
|
|
222
|
+
element_text = caption
|
|
223
|
+
else:
|
|
224
|
+
text = item.get("text", "")
|
|
225
|
+
if text:
|
|
226
|
+
markdown_blocks.append(text)
|
|
227
|
+
text_lines.append(text)
|
|
228
|
+
element_text = text
|
|
229
|
+
|
|
230
|
+
elements.append(
|
|
231
|
+
{
|
|
232
|
+
"kind": kind,
|
|
233
|
+
"text": element_text,
|
|
234
|
+
"bbox": item.get("bbox"),
|
|
235
|
+
"metadata": {"page_idx": page_idx},
|
|
236
|
+
}
|
|
237
|
+
)
|
|
238
|
+
|
|
239
|
+
return {
|
|
240
|
+
"page_number": page_idx + 1,
|
|
241
|
+
"markdown": "\n\n".join(markdown_blocks) or document_markdown,
|
|
242
|
+
"plain_text": "\n".join(text_lines),
|
|
243
|
+
"elements": elements,
|
|
244
|
+
"tables": tables,
|
|
245
|
+
"images": images,
|
|
246
|
+
"engine_specific": {"content_list": items},
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
def _join_caption(self, caption: Any) -> str:
|
|
250
|
+
if isinstance(caption, str):
|
|
251
|
+
return caption.strip()
|
|
252
|
+
if isinstance(caption, list):
|
|
253
|
+
return " ".join(str(part).strip() for part in caption if str(part).strip())
|
|
254
|
+
return ""
|
|
255
|
+
|
|
256
|
+
def _extract_markdown(self, response: dict[str, Any]) -> str:
|
|
257
|
+
candidates = [
|
|
258
|
+
response.get("md_content"),
|
|
259
|
+
response.get("markdown"),
|
|
260
|
+
response.get("md"),
|
|
261
|
+
response.get("full_md"),
|
|
262
|
+
]
|
|
263
|
+
result = response.get("result")
|
|
264
|
+
if isinstance(result, dict):
|
|
265
|
+
candidates.extend(
|
|
266
|
+
[
|
|
267
|
+
result.get("md_content"),
|
|
268
|
+
result.get("markdown"),
|
|
269
|
+
result.get("md"),
|
|
270
|
+
result.get("full_md"),
|
|
271
|
+
]
|
|
272
|
+
)
|
|
273
|
+
results = response.get("results")
|
|
274
|
+
if isinstance(results, dict):
|
|
275
|
+
for file_result in results.values():
|
|
276
|
+
if isinstance(file_result, dict):
|
|
277
|
+
candidates.extend(
|
|
278
|
+
[
|
|
279
|
+
file_result.get("md_content"),
|
|
280
|
+
file_result.get("markdown"),
|
|
281
|
+
file_result.get("md"),
|
|
282
|
+
file_result.get("full_md"),
|
|
283
|
+
]
|
|
284
|
+
)
|
|
285
|
+
for candidate in candidates:
|
|
286
|
+
if isinstance(candidate, str) and candidate:
|
|
287
|
+
return candidate
|
|
288
|
+
return ""
|
|
289
|
+
|
|
290
|
+
def _extract_content_list(self, response: dict[str, Any]) -> list[dict[str, Any]]:
|
|
291
|
+
for key in ("content_list", "content_list_v2"):
|
|
292
|
+
value = response.get(key)
|
|
293
|
+
if isinstance(value, str):
|
|
294
|
+
try:
|
|
295
|
+
value = json.loads(value)
|
|
296
|
+
except json.JSONDecodeError:
|
|
297
|
+
continue
|
|
298
|
+
if isinstance(value, list):
|
|
299
|
+
if value and isinstance(value[0], dict):
|
|
300
|
+
return value
|
|
301
|
+
if value and isinstance(value[0], list):
|
|
302
|
+
flattened = []
|
|
303
|
+
for page_idx, page_items in enumerate(value):
|
|
304
|
+
for item in page_items:
|
|
305
|
+
if isinstance(item, dict):
|
|
306
|
+
flattened.append({"page_idx": page_idx, **item})
|
|
307
|
+
return flattened
|
|
308
|
+
result = response.get("result")
|
|
309
|
+
if isinstance(result, dict):
|
|
310
|
+
return self._extract_content_list(result)
|
|
311
|
+
results = response.get("results")
|
|
312
|
+
if isinstance(results, dict):
|
|
313
|
+
for file_result in results.values():
|
|
314
|
+
if isinstance(file_result, dict):
|
|
315
|
+
content_list = self._extract_content_list(file_result)
|
|
316
|
+
if content_list:
|
|
317
|
+
return content_list
|
|
318
|
+
return []
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import shlex
|
|
6
|
+
import shutil
|
|
7
|
+
import socket
|
|
8
|
+
import subprocess
|
|
9
|
+
import sys
|
|
10
|
+
import tempfile
|
|
11
|
+
import time
|
|
12
|
+
from contextlib import contextmanager
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from langparse.engines.pdf.mineru_client import MinerUClient
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class MinerUServiceManager:
|
|
19
|
+
def __init__(
|
|
20
|
+
self,
|
|
21
|
+
api_url: str | None = None,
|
|
22
|
+
host: str = "127.0.0.1",
|
|
23
|
+
port: int = 8000,
|
|
24
|
+
command: str = "mineru-api",
|
|
25
|
+
start_timeout: float = 30.0,
|
|
26
|
+
request_timeout: float = 300.0,
|
|
27
|
+
model_dir: str | None = None,
|
|
28
|
+
download_dir: str | None = None,
|
|
29
|
+
model_policy: str = "download_if_missing",
|
|
30
|
+
model_source: str | None = None,
|
|
31
|
+
auto_install_runtime: bool = False,
|
|
32
|
+
runtime_package: str = "mineru>=3.4,<4",
|
|
33
|
+
):
|
|
34
|
+
self.api_url = api_url.rstrip("/") if api_url else None
|
|
35
|
+
self.host = host
|
|
36
|
+
self.port = port
|
|
37
|
+
self.command = command
|
|
38
|
+
self.start_timeout = start_timeout
|
|
39
|
+
self.request_timeout = request_timeout
|
|
40
|
+
self.model_dir = model_dir
|
|
41
|
+
self.download_dir = download_dir
|
|
42
|
+
self.model_policy = model_policy
|
|
43
|
+
self.model_source = model_source
|
|
44
|
+
self.auto_install_runtime = auto_install_runtime
|
|
45
|
+
self.runtime_package = runtime_package
|
|
46
|
+
|
|
47
|
+
@contextmanager
|
|
48
|
+
def running_service(self):
|
|
49
|
+
if self.api_url:
|
|
50
|
+
yield self.api_url
|
|
51
|
+
return
|
|
52
|
+
|
|
53
|
+
base_url = f"http://{self.host}:{self.port}"
|
|
54
|
+
client = MinerUClient(base_url, timeout=self.request_timeout)
|
|
55
|
+
if self._is_healthy(client):
|
|
56
|
+
yield base_url
|
|
57
|
+
return
|
|
58
|
+
|
|
59
|
+
self._validate_model_policy()
|
|
60
|
+
home_override_cm = self._prepare_local_home()
|
|
61
|
+
with home_override_cm as home_override:
|
|
62
|
+
process = self._start_local_service(home_override=home_override)
|
|
63
|
+
try:
|
|
64
|
+
self._wait_until_ready(client)
|
|
65
|
+
yield base_url
|
|
66
|
+
finally:
|
|
67
|
+
self._stop_process(process)
|
|
68
|
+
|
|
69
|
+
def _is_healthy(self, client: MinerUClient) -> bool:
|
|
70
|
+
try:
|
|
71
|
+
client.health()
|
|
72
|
+
except Exception:
|
|
73
|
+
return False
|
|
74
|
+
return True
|
|
75
|
+
|
|
76
|
+
def _start_local_service(self, home_override: str | None = None) -> subprocess.Popen:
|
|
77
|
+
args = shlex.split(self.command) + ["--host", self.host, "--port", str(self.port)]
|
|
78
|
+
env = self._build_process_env(home_override=home_override)
|
|
79
|
+
if not self._command_available(self.command):
|
|
80
|
+
if self.auto_install_runtime:
|
|
81
|
+
self._install_runtime()
|
|
82
|
+
else:
|
|
83
|
+
raise RuntimeError(
|
|
84
|
+
"Unable to start local mineru-api service using command: "
|
|
85
|
+
f"{self.command}. MinerU runtime was not found. Install it with "
|
|
86
|
+
'`pip install -U "mineru>=3.4,<4"` (plus the official backend extra '
|
|
87
|
+
"needed for local inference) or retry with auto_install_runtime=True."
|
|
88
|
+
)
|
|
89
|
+
try:
|
|
90
|
+
return subprocess.Popen(
|
|
91
|
+
args,
|
|
92
|
+
stdout=subprocess.DEVNULL,
|
|
93
|
+
stderr=subprocess.DEVNULL,
|
|
94
|
+
env=env,
|
|
95
|
+
)
|
|
96
|
+
except FileNotFoundError as exc:
|
|
97
|
+
raise RuntimeError(
|
|
98
|
+
f"Unable to start local mineru-api service using command: {self.command}"
|
|
99
|
+
) from exc
|
|
100
|
+
|
|
101
|
+
def _command_available(self, command: str) -> bool:
|
|
102
|
+
executable = shlex.split(command)[0]
|
|
103
|
+
return shutil.which(executable) is not None
|
|
104
|
+
|
|
105
|
+
def _install_runtime(self) -> None:
|
|
106
|
+
try:
|
|
107
|
+
subprocess.run(
|
|
108
|
+
[sys.executable, "-m", "pip", "install", "-U", self.runtime_package],
|
|
109
|
+
check=True,
|
|
110
|
+
)
|
|
111
|
+
except subprocess.CalledProcessError as exc:
|
|
112
|
+
raise RuntimeError(
|
|
113
|
+
f"Failed to install MinerU runtime package: {self.runtime_package}"
|
|
114
|
+
) from exc
|
|
115
|
+
|
|
116
|
+
def _wait_until_ready(self, client: MinerUClient) -> None:
|
|
117
|
+
deadline = time.time() + self.start_timeout
|
|
118
|
+
last_error: Exception | None = None
|
|
119
|
+
while time.time() < deadline:
|
|
120
|
+
try:
|
|
121
|
+
client.health()
|
|
122
|
+
return
|
|
123
|
+
except Exception as exc: # pragma: no cover - exercised in timeout flow
|
|
124
|
+
last_error = exc
|
|
125
|
+
time.sleep(0.25)
|
|
126
|
+
raise RuntimeError(
|
|
127
|
+
"Timed out waiting for local mineru-api service to become ready."
|
|
128
|
+
) from last_error
|
|
129
|
+
|
|
130
|
+
def _stop_process(self, process: subprocess.Popen) -> None:
|
|
131
|
+
if process.poll() is not None:
|
|
132
|
+
return
|
|
133
|
+
process.terminate()
|
|
134
|
+
try:
|
|
135
|
+
process.wait(timeout=5)
|
|
136
|
+
except subprocess.TimeoutExpired:
|
|
137
|
+
process.kill()
|
|
138
|
+
process.wait(timeout=5)
|
|
139
|
+
|
|
140
|
+
@staticmethod
|
|
141
|
+
def find_free_port() -> int:
|
|
142
|
+
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock:
|
|
143
|
+
sock.bind(("127.0.0.1", 0))
|
|
144
|
+
return int(sock.getsockname()[1])
|
|
145
|
+
|
|
146
|
+
def _validate_model_policy(self) -> None:
|
|
147
|
+
if self.model_policy not in {"download_if_missing", "require_existing"}:
|
|
148
|
+
raise ValueError(
|
|
149
|
+
"Unsupported MinerU model_policy: "
|
|
150
|
+
f"{self.model_policy}. Expected 'download_if_missing' or 'require_existing'."
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
if self.model_policy != "require_existing":
|
|
154
|
+
return
|
|
155
|
+
|
|
156
|
+
if self.model_dir:
|
|
157
|
+
model_path = Path(self.model_dir).expanduser()
|
|
158
|
+
if not model_path.exists() or not any(model_path.iterdir()):
|
|
159
|
+
raise RuntimeError(
|
|
160
|
+
f"MinerU model_policy=require_existing but model_dir is missing or empty: {model_path}"
|
|
161
|
+
)
|
|
162
|
+
return
|
|
163
|
+
|
|
164
|
+
home_root = self._resolve_home_root()
|
|
165
|
+
mineru_config = home_root / "mineru.json"
|
|
166
|
+
mineru_cache = home_root / ".mineru"
|
|
167
|
+
if not mineru_config.exists() and not mineru_cache.exists():
|
|
168
|
+
raise RuntimeError(
|
|
169
|
+
"MinerU model_policy=require_existing requires either model_dir or an existing "
|
|
170
|
+
f"local MinerU home with mineru.json/.mineru under: {home_root}"
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
def _resolve_home_root(self) -> Path:
|
|
174
|
+
if self.download_dir:
|
|
175
|
+
return Path(self.download_dir).expanduser()
|
|
176
|
+
return Path.home()
|
|
177
|
+
|
|
178
|
+
def _build_process_env(self, home_override: str | None = None) -> dict[str, str]:
|
|
179
|
+
env = os.environ.copy()
|
|
180
|
+
if home_override:
|
|
181
|
+
env["HOME"] = home_override
|
|
182
|
+
|
|
183
|
+
effective_model_source = self.model_source
|
|
184
|
+
if effective_model_source is None and self.model_dir:
|
|
185
|
+
effective_model_source = "local"
|
|
186
|
+
|
|
187
|
+
if effective_model_source:
|
|
188
|
+
env["MINERU_MODEL_SOURCE"] = effective_model_source
|
|
189
|
+
|
|
190
|
+
return env
|
|
191
|
+
|
|
192
|
+
@contextmanager
|
|
193
|
+
def _prepare_local_home(self):
|
|
194
|
+
if self.model_dir:
|
|
195
|
+
configured_root = Path(self.download_dir).expanduser() if self.download_dir else None
|
|
196
|
+
if configured_root is not None:
|
|
197
|
+
configured_root.mkdir(parents=True, exist_ok=True)
|
|
198
|
+
self._write_mineru_config(configured_root)
|
|
199
|
+
yield str(configured_root)
|
|
200
|
+
return
|
|
201
|
+
|
|
202
|
+
with tempfile.TemporaryDirectory(prefix="langparse-mineru-home-") as temp_home:
|
|
203
|
+
temp_root = Path(temp_home)
|
|
204
|
+
self._write_mineru_config(temp_root)
|
|
205
|
+
yield temp_home
|
|
206
|
+
return
|
|
207
|
+
|
|
208
|
+
if self.download_dir:
|
|
209
|
+
configured_root = Path(self.download_dir).expanduser()
|
|
210
|
+
configured_root.mkdir(parents=True, exist_ok=True)
|
|
211
|
+
yield str(configured_root)
|
|
212
|
+
return
|
|
213
|
+
|
|
214
|
+
yield None
|
|
215
|
+
|
|
216
|
+
def _write_mineru_config(self, home_root: Path) -> None:
|
|
217
|
+
model_path = str(Path(self.model_dir).expanduser())
|
|
218
|
+
config_path = home_root / "mineru.json"
|
|
219
|
+
config = {
|
|
220
|
+
"models-dir": {
|
|
221
|
+
"pipeline": model_path,
|
|
222
|
+
"vlm": model_path,
|
|
223
|
+
}
|
|
224
|
+
}
|
|
225
|
+
config_path.write_text(json.dumps(config, ensure_ascii=False, indent=2), encoding="utf-8")
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""
|
|
2
|
+
OCR fallback for PDF pages whose content is an image rather than text.
|
|
3
|
+
|
|
4
|
+
The dangerous case is not an empty page -- it is a scanned page carrying a
|
|
5
|
+
watermark, which leaves just enough extracted text that the parse reports
|
|
6
|
+
success while the actual content is never read. Detection therefore looks for
|
|
7
|
+
"almost no text next to an image", not "no text".
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from collections.abc import Callable
|
|
13
|
+
from typing import Any, Protocol
|
|
14
|
+
|
|
15
|
+
#: A full page of body text runs well over a thousand characters. This is the
|
|
16
|
+
#: ceiling below which a page covered by an image is treated as a scan rather
|
|
17
|
+
#: than as text -- data/domain/scan.pdf carries 145 characters of watermark.
|
|
18
|
+
DEFAULT_MIN_CHARS = 500
|
|
19
|
+
#: Fraction of the page an image must cover before the page counts as scanned.
|
|
20
|
+
DEFAULT_MIN_IMAGE_COVERAGE = 0.5
|
|
21
|
+
DEFAULT_RESOLUTION = 150
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class Recogniser(Protocol):
|
|
25
|
+
"""Matches rapidocr_onnxruntime's RapidOCR call signature."""
|
|
26
|
+
|
|
27
|
+
def __call__(self, image: Any) -> tuple[Any, Any]: ...
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def image_coverage(page) -> float:
|
|
31
|
+
"""Largest single image's area as a fraction of the page."""
|
|
32
|
+
images = getattr(page, "images", None)
|
|
33
|
+
if not images:
|
|
34
|
+
return 0.0
|
|
35
|
+
|
|
36
|
+
page_area = (getattr(page, "width", 0) or 0) * (getattr(page, "height", 0) or 0)
|
|
37
|
+
if page_area <= 0:
|
|
38
|
+
return 0.0
|
|
39
|
+
|
|
40
|
+
largest = 0.0
|
|
41
|
+
for image in images:
|
|
42
|
+
width = abs(float(image.get("x1", 0)) - float(image.get("x0", 0)))
|
|
43
|
+
height = abs(float(image.get("bottom", 0)) - float(image.get("top", 0)))
|
|
44
|
+
largest = max(largest, width * height)
|
|
45
|
+
return min(1.0, largest / page_area)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def needs_ocr(
|
|
49
|
+
page,
|
|
50
|
+
min_chars: int = DEFAULT_MIN_CHARS,
|
|
51
|
+
min_image_coverage: float = DEFAULT_MIN_IMAGE_COVERAGE,
|
|
52
|
+
) -> bool:
|
|
53
|
+
"""
|
|
54
|
+
Whether a page's text layer is too thin to trust.
|
|
55
|
+
|
|
56
|
+
Text length alone is not enough: data/domain/scan.pdf carries 145 characters
|
|
57
|
+
of rotated watermark per page, which clears any threshold low enough to
|
|
58
|
+
avoid firing on genuinely sparse text pages. The page-covering image is the
|
|
59
|
+
signal that distinguishes them, so both conditions must hold.
|
|
60
|
+
"""
|
|
61
|
+
if image_coverage(page) < min_image_coverage:
|
|
62
|
+
return False
|
|
63
|
+
|
|
64
|
+
text = page.extract_text() or ""
|
|
65
|
+
return len(text.strip()) < min_chars
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def ocr_page_text(
|
|
69
|
+
page,
|
|
70
|
+
recogniser: Recogniser,
|
|
71
|
+
resolution: int = DEFAULT_RESOLUTION,
|
|
72
|
+
) -> str:
|
|
73
|
+
"""Rasterise a page and return the recognised text, one line per detection."""
|
|
74
|
+
image = page.to_image(resolution=resolution).original
|
|
75
|
+
result, _elapsed = recogniser(image)
|
|
76
|
+
if not result:
|
|
77
|
+
return ""
|
|
78
|
+
|
|
79
|
+
lines = []
|
|
80
|
+
for detection in result:
|
|
81
|
+
# rapidocr yields [box, text, score]; be tolerant of shape changes.
|
|
82
|
+
if isinstance(detection, (list, tuple)) and len(detection) >= 2:
|
|
83
|
+
text = detection[1]
|
|
84
|
+
else:
|
|
85
|
+
text = detection
|
|
86
|
+
if isinstance(text, str) and text.strip():
|
|
87
|
+
lines.append(text.strip())
|
|
88
|
+
return "\n".join(lines)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def load_recogniser() -> Callable[[Any], tuple[Any, Any]]:
|
|
92
|
+
"""Build the default rapidocr recogniser, with an actionable error if absent."""
|
|
93
|
+
try:
|
|
94
|
+
from rapidocr_onnxruntime import RapidOCR
|
|
95
|
+
except ImportError as exc:
|
|
96
|
+
raise ImportError(
|
|
97
|
+
"OCR fallback needs rapidocr_onnxruntime. Install it with "
|
|
98
|
+
'`pip install "langparse[ocr]"`, or pass enable_ocr=False to skip it.'
|
|
99
|
+
) from exc
|
|
100
|
+
|
|
101
|
+
return RapidOCR()
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
from collections.abc import Iterator
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
from langparse.core.engine import PageResult
|
|
5
|
+
from langparse.engines.pdf.simple import BasePDFEngine
|
|
6
|
+
from langparse.logging import get_logger
|
|
7
|
+
|
|
8
|
+
logger = get_logger(__name__)
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class PaddleOCRVLEngine(BasePDFEngine):
|
|
12
|
+
"""
|
|
13
|
+
Adapter for PaddleOCR + Layout Analysis or PP-Structure.
|
|
14
|
+
Can be local or via API.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
def process(self, file_path: Path, **kwargs) -> Iterator[PageResult]:
|
|
18
|
+
logger.debug("PaddleOCR processing %s", file_path)
|
|
19
|
+
# TODO: Integrate PaddleOCR / PP-Structure logic
|
|
20
|
+
raise NotImplementedError("PaddleOCR integration is pending.")
|