langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,318 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import mimetypes
5
+ import uuid
6
+ from html.parser import HTMLParser
7
+ from pathlib import Path
8
+ from typing import Any
9
+ from urllib import request
10
+ from urllib.error import HTTPError, URLError
11
+
12
+
13
+ class _HTMLTableReader(HTMLParser):
14
+ """Collects rows/cells from the HTML fragment MinerU returns as ``table_body``."""
15
+
16
+ def __init__(self):
17
+ super().__init__(convert_charrefs=True)
18
+ self.rows: list[list[str]] = []
19
+ self._row: list[str] | None = None
20
+ self._cell: list[str] | None = None
21
+
22
+ def handle_starttag(self, tag, attrs):
23
+ if tag == "tr":
24
+ self._row = []
25
+ elif tag in ("td", "th"):
26
+ self._cell = []
27
+
28
+ def handle_endtag(self, tag):
29
+ if tag in ("td", "th") and self._cell is not None:
30
+ text = " ".join("".join(self._cell).split())
31
+ if self._row is None:
32
+ self._row = []
33
+ self._row.append(text)
34
+ self._cell = None
35
+ elif tag == "tr" and self._row is not None:
36
+ self.rows.append(self._row)
37
+ self._row = None
38
+
39
+ def handle_data(self, data):
40
+ if self._cell is not None:
41
+ self._cell.append(data)
42
+
43
+
44
+ def parse_html_table(markup: str) -> list[list[str]]:
45
+ """Parse an HTML table fragment into rows of cell text. Returns [] if unparseable."""
46
+ if not markup:
47
+ return []
48
+ reader = _HTMLTableReader()
49
+ try:
50
+ reader.feed(markup)
51
+ reader.close()
52
+ except Exception:
53
+ return []
54
+ return [row for row in reader.rows if row]
55
+
56
+
57
+ def rows_to_markdown(rows: list[list[str]]) -> str:
58
+ """Render rows as a Markdown table, padding short rows to the header width."""
59
+ if not rows:
60
+ return ""
61
+ width = max(len(row) for row in rows)
62
+ padded = [row + [""] * (width - len(row)) for row in rows]
63
+ lines = [f"| {' | '.join(padded[0])} |", f"| {' | '.join(['---'] * width)} |"]
64
+ lines.extend(f"| {' | '.join(row)} |" for row in padded[1:])
65
+ return "\n".join(lines)
66
+
67
+
68
+ class MinerUClient:
69
+ def __init__(self, base_url: str, timeout: float = 300.0):
70
+ self.base_url = base_url.rstrip("/")
71
+ self.timeout = timeout
72
+
73
+ def health(self) -> dict[str, Any]:
74
+ return self._request_json("GET", "/health")
75
+
76
+ def parse_file(self, file_path: Path, runtime_config: dict[str, Any]) -> list[dict[str, Any]]:
77
+ response = self._request_json(
78
+ "POST",
79
+ "/file_parse",
80
+ fields=self._build_form_fields(runtime_config),
81
+ file_path=file_path,
82
+ )
83
+ return self._normalize_parse_response(response)
84
+
85
+ def _build_form_fields(self, runtime_config: dict[str, Any]) -> dict[str, str]:
86
+ fields = {
87
+ "return_md": "true",
88
+ "return_content_list": "true",
89
+ "response_format_zip": "false",
90
+ }
91
+ extra_options = runtime_config.get("extra_options", {})
92
+ if runtime_config.get("enable_ocr") is False:
93
+ fields["parse_method"] = "txt"
94
+ if runtime_config.get("device"):
95
+ fields["device"] = str(runtime_config["device"])
96
+ if runtime_config.get("model_dir"):
97
+ fields["model_dir"] = str(runtime_config["model_dir"])
98
+ if runtime_config.get("download_dir"):
99
+ fields["download_dir"] = str(runtime_config["download_dir"])
100
+ for key, value in extra_options.items():
101
+ if value is None:
102
+ continue
103
+ fields[str(key)] = str(value)
104
+ return fields
105
+
106
+ def _request_json(
107
+ self,
108
+ method: str,
109
+ path: str,
110
+ fields: dict[str, str] | None = None,
111
+ file_path: Path | None = None,
112
+ ) -> dict[str, Any]:
113
+ headers = {"Accept": "application/json"}
114
+ data = None
115
+ if file_path is not None:
116
+ data, content_type = self._encode_multipart_form(fields or {}, file_path)
117
+ headers["Content-Type"] = content_type
118
+
119
+ req = request.Request(
120
+ f"{self.base_url}{path}",
121
+ data=data,
122
+ headers=headers,
123
+ method=method,
124
+ )
125
+ try:
126
+ with request.urlopen(req, timeout=self.timeout) as response:
127
+ payload = response.read().decode("utf-8")
128
+ except HTTPError as exc:
129
+ detail = exc.read().decode("utf-8", errors="replace")
130
+ raise RuntimeError(f"MinerU API request failed with HTTP {exc.code}: {detail}") from exc
131
+ except URLError as exc:
132
+ raise RuntimeError(f"MinerU API request failed: {exc.reason}") from exc
133
+
134
+ try:
135
+ return json.loads(payload)
136
+ except json.JSONDecodeError as exc:
137
+ raise RuntimeError("MinerU API returned a non-JSON response.") from exc
138
+
139
+ def _encode_multipart_form(self, fields: dict[str, str], file_path: Path) -> tuple[bytes, str]:
140
+ boundary = f"----langparse-mineru-{uuid.uuid4().hex}"
141
+ lines: list[bytes] = []
142
+ for name, value in fields.items():
143
+ lines.extend(
144
+ [
145
+ f"--{boundary}".encode(),
146
+ f'Content-Disposition: form-data; name="{name}"'.encode(),
147
+ b"",
148
+ str(value).encode("utf-8"),
149
+ ]
150
+ )
151
+
152
+ content_type = mimetypes.guess_type(file_path.name)[0] or "application/octet-stream"
153
+ file_bytes = file_path.read_bytes()
154
+ lines.extend(
155
+ [
156
+ f"--{boundary}".encode(),
157
+ f'Content-Disposition: form-data; name="files"; filename="{file_path.name}"'.encode(),
158
+ f"Content-Type: {content_type}".encode(),
159
+ b"",
160
+ file_bytes,
161
+ f"--{boundary}--".encode(),
162
+ b"",
163
+ ]
164
+ )
165
+ return b"\r\n".join(lines), f"multipart/form-data; boundary={boundary}"
166
+
167
+ def _normalize_parse_response(self, response: dict[str, Any]) -> list[dict[str, Any]]:
168
+ markdown = self._extract_markdown(response)
169
+ content_list = self._extract_content_list(response)
170
+ if not content_list:
171
+ return [{"page_number": 1, "markdown": markdown}]
172
+
173
+ page_map: dict[int, list[dict[str, Any]]] = {}
174
+ for item in content_list:
175
+ page_idx = int(item.get("page_idx", 0))
176
+ page_map.setdefault(page_idx, []).append(item)
177
+
178
+ pages = []
179
+ for page_idx in sorted(page_map):
180
+ items = page_map[page_idx]
181
+ pages.append(self._build_page(page_idx, items, markdown))
182
+ return pages
183
+
184
+ def _build_page(
185
+ self, page_idx: int, items: list[dict[str, Any]], document_markdown: str
186
+ ) -> dict[str, Any]:
187
+ markdown_blocks: list[str] = []
188
+ text_lines: list[str] = []
189
+ tables: list[dict[str, Any]] = []
190
+ images: list[dict[str, Any]] = []
191
+ elements: list[dict[str, Any]] = []
192
+
193
+ for item in items:
194
+ kind = item.get("type", "text")
195
+ caption = self._join_caption(item.get(f"{kind}_caption"))
196
+
197
+ if kind == "table":
198
+ rows = parse_html_table(item.get("table_body", ""))
199
+ table_markdown = rows_to_markdown(rows)
200
+ tables.append(
201
+ {
202
+ "rows": rows,
203
+ "caption": caption,
204
+ "html": item.get("table_body", ""),
205
+ "img_path": item.get("img_path"),
206
+ }
207
+ )
208
+ block = "\n\n".join(part for part in (caption, table_markdown) if part)
209
+ if block:
210
+ markdown_blocks.append(block)
211
+ element_text = table_markdown
212
+ elif kind == "image":
213
+ images.append(
214
+ {
215
+ "path": item.get("img_path"),
216
+ "caption": caption,
217
+ "footnote": self._join_caption(item.get("image_footnote")),
218
+ }
219
+ )
220
+ block = f"![{caption}]({item.get('img_path') or ''})"
221
+ markdown_blocks.append(block)
222
+ element_text = caption
223
+ else:
224
+ text = item.get("text", "")
225
+ if text:
226
+ markdown_blocks.append(text)
227
+ text_lines.append(text)
228
+ element_text = text
229
+
230
+ elements.append(
231
+ {
232
+ "kind": kind,
233
+ "text": element_text,
234
+ "bbox": item.get("bbox"),
235
+ "metadata": {"page_idx": page_idx},
236
+ }
237
+ )
238
+
239
+ return {
240
+ "page_number": page_idx + 1,
241
+ "markdown": "\n\n".join(markdown_blocks) or document_markdown,
242
+ "plain_text": "\n".join(text_lines),
243
+ "elements": elements,
244
+ "tables": tables,
245
+ "images": images,
246
+ "engine_specific": {"content_list": items},
247
+ }
248
+
249
+ def _join_caption(self, caption: Any) -> str:
250
+ if isinstance(caption, str):
251
+ return caption.strip()
252
+ if isinstance(caption, list):
253
+ return " ".join(str(part).strip() for part in caption if str(part).strip())
254
+ return ""
255
+
256
+ def _extract_markdown(self, response: dict[str, Any]) -> str:
257
+ candidates = [
258
+ response.get("md_content"),
259
+ response.get("markdown"),
260
+ response.get("md"),
261
+ response.get("full_md"),
262
+ ]
263
+ result = response.get("result")
264
+ if isinstance(result, dict):
265
+ candidates.extend(
266
+ [
267
+ result.get("md_content"),
268
+ result.get("markdown"),
269
+ result.get("md"),
270
+ result.get("full_md"),
271
+ ]
272
+ )
273
+ results = response.get("results")
274
+ if isinstance(results, dict):
275
+ for file_result in results.values():
276
+ if isinstance(file_result, dict):
277
+ candidates.extend(
278
+ [
279
+ file_result.get("md_content"),
280
+ file_result.get("markdown"),
281
+ file_result.get("md"),
282
+ file_result.get("full_md"),
283
+ ]
284
+ )
285
+ for candidate in candidates:
286
+ if isinstance(candidate, str) and candidate:
287
+ return candidate
288
+ return ""
289
+
290
+ def _extract_content_list(self, response: dict[str, Any]) -> list[dict[str, Any]]:
291
+ for key in ("content_list", "content_list_v2"):
292
+ value = response.get(key)
293
+ if isinstance(value, str):
294
+ try:
295
+ value = json.loads(value)
296
+ except json.JSONDecodeError:
297
+ continue
298
+ if isinstance(value, list):
299
+ if value and isinstance(value[0], dict):
300
+ return value
301
+ if value and isinstance(value[0], list):
302
+ flattened = []
303
+ for page_idx, page_items in enumerate(value):
304
+ for item in page_items:
305
+ if isinstance(item, dict):
306
+ flattened.append({"page_idx": page_idx, **item})
307
+ return flattened
308
+ result = response.get("result")
309
+ if isinstance(result, dict):
310
+ return self._extract_content_list(result)
311
+ results = response.get("results")
312
+ if isinstance(results, dict):
313
+ for file_result in results.values():
314
+ if isinstance(file_result, dict):
315
+ content_list = self._extract_content_list(file_result)
316
+ if content_list:
317
+ return content_list
318
+ return []
@@ -0,0 +1,225 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import os
5
+ import shlex
6
+ import shutil
7
+ import socket
8
+ import subprocess
9
+ import sys
10
+ import tempfile
11
+ import time
12
+ from contextlib import contextmanager
13
+ from pathlib import Path
14
+
15
+ from langparse.engines.pdf.mineru_client import MinerUClient
16
+
17
+
18
+ class MinerUServiceManager:
19
+ def __init__(
20
+ self,
21
+ api_url: str | None = None,
22
+ host: str = "127.0.0.1",
23
+ port: int = 8000,
24
+ command: str = "mineru-api",
25
+ start_timeout: float = 30.0,
26
+ request_timeout: float = 300.0,
27
+ model_dir: str | None = None,
28
+ download_dir: str | None = None,
29
+ model_policy: str = "download_if_missing",
30
+ model_source: str | None = None,
31
+ auto_install_runtime: bool = False,
32
+ runtime_package: str = "mineru>=3.4,<4",
33
+ ):
34
+ self.api_url = api_url.rstrip("/") if api_url else None
35
+ self.host = host
36
+ self.port = port
37
+ self.command = command
38
+ self.start_timeout = start_timeout
39
+ self.request_timeout = request_timeout
40
+ self.model_dir = model_dir
41
+ self.download_dir = download_dir
42
+ self.model_policy = model_policy
43
+ self.model_source = model_source
44
+ self.auto_install_runtime = auto_install_runtime
45
+ self.runtime_package = runtime_package
46
+
47
+ @contextmanager
48
+ def running_service(self):
49
+ if self.api_url:
50
+ yield self.api_url
51
+ return
52
+
53
+ base_url = f"http://{self.host}:{self.port}"
54
+ client = MinerUClient(base_url, timeout=self.request_timeout)
55
+ if self._is_healthy(client):
56
+ yield base_url
57
+ return
58
+
59
+ self._validate_model_policy()
60
+ home_override_cm = self._prepare_local_home()
61
+ with home_override_cm as home_override:
62
+ process = self._start_local_service(home_override=home_override)
63
+ try:
64
+ self._wait_until_ready(client)
65
+ yield base_url
66
+ finally:
67
+ self._stop_process(process)
68
+
69
+ def _is_healthy(self, client: MinerUClient) -> bool:
70
+ try:
71
+ client.health()
72
+ except Exception:
73
+ return False
74
+ return True
75
+
76
+ def _start_local_service(self, home_override: str | None = None) -> subprocess.Popen:
77
+ args = shlex.split(self.command) + ["--host", self.host, "--port", str(self.port)]
78
+ env = self._build_process_env(home_override=home_override)
79
+ if not self._command_available(self.command):
80
+ if self.auto_install_runtime:
81
+ self._install_runtime()
82
+ else:
83
+ raise RuntimeError(
84
+ "Unable to start local mineru-api service using command: "
85
+ f"{self.command}. MinerU runtime was not found. Install it with "
86
+ '`pip install -U "mineru>=3.4,<4"` (plus the official backend extra '
87
+ "needed for local inference) or retry with auto_install_runtime=True."
88
+ )
89
+ try:
90
+ return subprocess.Popen(
91
+ args,
92
+ stdout=subprocess.DEVNULL,
93
+ stderr=subprocess.DEVNULL,
94
+ env=env,
95
+ )
96
+ except FileNotFoundError as exc:
97
+ raise RuntimeError(
98
+ f"Unable to start local mineru-api service using command: {self.command}"
99
+ ) from exc
100
+
101
+ def _command_available(self, command: str) -> bool:
102
+ executable = shlex.split(command)[0]
103
+ return shutil.which(executable) is not None
104
+
105
+ def _install_runtime(self) -> None:
106
+ try:
107
+ subprocess.run(
108
+ [sys.executable, "-m", "pip", "install", "-U", self.runtime_package],
109
+ check=True,
110
+ )
111
+ except subprocess.CalledProcessError as exc:
112
+ raise RuntimeError(
113
+ f"Failed to install MinerU runtime package: {self.runtime_package}"
114
+ ) from exc
115
+
116
+ def _wait_until_ready(self, client: MinerUClient) -> None:
117
+ deadline = time.time() + self.start_timeout
118
+ last_error: Exception | None = None
119
+ while time.time() < deadline:
120
+ try:
121
+ client.health()
122
+ return
123
+ except Exception as exc: # pragma: no cover - exercised in timeout flow
124
+ last_error = exc
125
+ time.sleep(0.25)
126
+ raise RuntimeError(
127
+ "Timed out waiting for local mineru-api service to become ready."
128
+ ) from last_error
129
+
130
+ def _stop_process(self, process: subprocess.Popen) -> None:
131
+ if process.poll() is not None:
132
+ return
133
+ process.terminate()
134
+ try:
135
+ process.wait(timeout=5)
136
+ except subprocess.TimeoutExpired:
137
+ process.kill()
138
+ process.wait(timeout=5)
139
+
140
+ @staticmethod
141
+ def find_free_port() -> int:
142
+ with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock:
143
+ sock.bind(("127.0.0.1", 0))
144
+ return int(sock.getsockname()[1])
145
+
146
+ def _validate_model_policy(self) -> None:
147
+ if self.model_policy not in {"download_if_missing", "require_existing"}:
148
+ raise ValueError(
149
+ "Unsupported MinerU model_policy: "
150
+ f"{self.model_policy}. Expected 'download_if_missing' or 'require_existing'."
151
+ )
152
+
153
+ if self.model_policy != "require_existing":
154
+ return
155
+
156
+ if self.model_dir:
157
+ model_path = Path(self.model_dir).expanduser()
158
+ if not model_path.exists() or not any(model_path.iterdir()):
159
+ raise RuntimeError(
160
+ f"MinerU model_policy=require_existing but model_dir is missing or empty: {model_path}"
161
+ )
162
+ return
163
+
164
+ home_root = self._resolve_home_root()
165
+ mineru_config = home_root / "mineru.json"
166
+ mineru_cache = home_root / ".mineru"
167
+ if not mineru_config.exists() and not mineru_cache.exists():
168
+ raise RuntimeError(
169
+ "MinerU model_policy=require_existing requires either model_dir or an existing "
170
+ f"local MinerU home with mineru.json/.mineru under: {home_root}"
171
+ )
172
+
173
+ def _resolve_home_root(self) -> Path:
174
+ if self.download_dir:
175
+ return Path(self.download_dir).expanduser()
176
+ return Path.home()
177
+
178
+ def _build_process_env(self, home_override: str | None = None) -> dict[str, str]:
179
+ env = os.environ.copy()
180
+ if home_override:
181
+ env["HOME"] = home_override
182
+
183
+ effective_model_source = self.model_source
184
+ if effective_model_source is None and self.model_dir:
185
+ effective_model_source = "local"
186
+
187
+ if effective_model_source:
188
+ env["MINERU_MODEL_SOURCE"] = effective_model_source
189
+
190
+ return env
191
+
192
+ @contextmanager
193
+ def _prepare_local_home(self):
194
+ if self.model_dir:
195
+ configured_root = Path(self.download_dir).expanduser() if self.download_dir else None
196
+ if configured_root is not None:
197
+ configured_root.mkdir(parents=True, exist_ok=True)
198
+ self._write_mineru_config(configured_root)
199
+ yield str(configured_root)
200
+ return
201
+
202
+ with tempfile.TemporaryDirectory(prefix="langparse-mineru-home-") as temp_home:
203
+ temp_root = Path(temp_home)
204
+ self._write_mineru_config(temp_root)
205
+ yield temp_home
206
+ return
207
+
208
+ if self.download_dir:
209
+ configured_root = Path(self.download_dir).expanduser()
210
+ configured_root.mkdir(parents=True, exist_ok=True)
211
+ yield str(configured_root)
212
+ return
213
+
214
+ yield None
215
+
216
+ def _write_mineru_config(self, home_root: Path) -> None:
217
+ model_path = str(Path(self.model_dir).expanduser())
218
+ config_path = home_root / "mineru.json"
219
+ config = {
220
+ "models-dir": {
221
+ "pipeline": model_path,
222
+ "vlm": model_path,
223
+ }
224
+ }
225
+ config_path.write_text(json.dumps(config, ensure_ascii=False, indent=2), encoding="utf-8")
@@ -0,0 +1,101 @@
1
+ """
2
+ OCR fallback for PDF pages whose content is an image rather than text.
3
+
4
+ The dangerous case is not an empty page -- it is a scanned page carrying a
5
+ watermark, which leaves just enough extracted text that the parse reports
6
+ success while the actual content is never read. Detection therefore looks for
7
+ "almost no text next to an image", not "no text".
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from collections.abc import Callable
13
+ from typing import Any, Protocol
14
+
15
+ #: A full page of body text runs well over a thousand characters. This is the
16
+ #: ceiling below which a page covered by an image is treated as a scan rather
17
+ #: than as text -- data/domain/scan.pdf carries 145 characters of watermark.
18
+ DEFAULT_MIN_CHARS = 500
19
+ #: Fraction of the page an image must cover before the page counts as scanned.
20
+ DEFAULT_MIN_IMAGE_COVERAGE = 0.5
21
+ DEFAULT_RESOLUTION = 150
22
+
23
+
24
+ class Recogniser(Protocol):
25
+ """Matches rapidocr_onnxruntime's RapidOCR call signature."""
26
+
27
+ def __call__(self, image: Any) -> tuple[Any, Any]: ...
28
+
29
+
30
+ def image_coverage(page) -> float:
31
+ """Largest single image's area as a fraction of the page."""
32
+ images = getattr(page, "images", None)
33
+ if not images:
34
+ return 0.0
35
+
36
+ page_area = (getattr(page, "width", 0) or 0) * (getattr(page, "height", 0) or 0)
37
+ if page_area <= 0:
38
+ return 0.0
39
+
40
+ largest = 0.0
41
+ for image in images:
42
+ width = abs(float(image.get("x1", 0)) - float(image.get("x0", 0)))
43
+ height = abs(float(image.get("bottom", 0)) - float(image.get("top", 0)))
44
+ largest = max(largest, width * height)
45
+ return min(1.0, largest / page_area)
46
+
47
+
48
+ def needs_ocr(
49
+ page,
50
+ min_chars: int = DEFAULT_MIN_CHARS,
51
+ min_image_coverage: float = DEFAULT_MIN_IMAGE_COVERAGE,
52
+ ) -> bool:
53
+ """
54
+ Whether a page's text layer is too thin to trust.
55
+
56
+ Text length alone is not enough: data/domain/scan.pdf carries 145 characters
57
+ of rotated watermark per page, which clears any threshold low enough to
58
+ avoid firing on genuinely sparse text pages. The page-covering image is the
59
+ signal that distinguishes them, so both conditions must hold.
60
+ """
61
+ if image_coverage(page) < min_image_coverage:
62
+ return False
63
+
64
+ text = page.extract_text() or ""
65
+ return len(text.strip()) < min_chars
66
+
67
+
68
+ def ocr_page_text(
69
+ page,
70
+ recogniser: Recogniser,
71
+ resolution: int = DEFAULT_RESOLUTION,
72
+ ) -> str:
73
+ """Rasterise a page and return the recognised text, one line per detection."""
74
+ image = page.to_image(resolution=resolution).original
75
+ result, _elapsed = recogniser(image)
76
+ if not result:
77
+ return ""
78
+
79
+ lines = []
80
+ for detection in result:
81
+ # rapidocr yields [box, text, score]; be tolerant of shape changes.
82
+ if isinstance(detection, (list, tuple)) and len(detection) >= 2:
83
+ text = detection[1]
84
+ else:
85
+ text = detection
86
+ if isinstance(text, str) and text.strip():
87
+ lines.append(text.strip())
88
+ return "\n".join(lines)
89
+
90
+
91
+ def load_recogniser() -> Callable[[Any], tuple[Any, Any]]:
92
+ """Build the default rapidocr recogniser, with an actionable error if absent."""
93
+ try:
94
+ from rapidocr_onnxruntime import RapidOCR
95
+ except ImportError as exc:
96
+ raise ImportError(
97
+ "OCR fallback needs rapidocr_onnxruntime. Install it with "
98
+ '`pip install "langparse[ocr]"`, or pass enable_ocr=False to skip it.'
99
+ ) from exc
100
+
101
+ return RapidOCR()
@@ -0,0 +1,20 @@
1
+ from collections.abc import Iterator
2
+ from pathlib import Path
3
+
4
+ from langparse.core.engine import PageResult
5
+ from langparse.engines.pdf.simple import BasePDFEngine
6
+ from langparse.logging import get_logger
7
+
8
+ logger = get_logger(__name__)
9
+
10
+
11
+ class PaddleOCRVLEngine(BasePDFEngine):
12
+ """
13
+ Adapter for PaddleOCR + Layout Analysis or PP-Structure.
14
+ Can be local or via API.
15
+ """
16
+
17
+ def process(self, file_path: Path, **kwargs) -> Iterator[PageResult]:
18
+ logger.debug("PaddleOCR processing %s", file_path)
19
+ # TODO: Integrate PaddleOCR / PP-Structure logic
20
+ raise NotImplementedError("PaddleOCR integration is pending.")