langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,86 @@
1
+ from __future__ import annotations
2
+
3
+ from collections import Counter
4
+ from collections.abc import Iterable
5
+ from pathlib import Path
6
+
7
+
8
+ def output_filename(source, fmt: str) -> str:
9
+ suffix = ".md" if fmt == "markdown" else ".json"
10
+ return f"{Path(source).stem}{suffix}"
11
+
12
+
13
+ def extension_tagged_filename(source, fmt: str) -> str:
14
+ """``report.docx`` -> ``report-docx.md``, to tell same-stem siblings apart."""
15
+ source_path = Path(source)
16
+ suffix = ".md" if fmt == "markdown" else ".json"
17
+ source_kind = source_path.suffix.lower().lstrip(".")
18
+ stem = f"{source_path.stem}-{source_kind}" if source_kind else source_path.stem
19
+ return f"{stem}{suffix}"
20
+
21
+
22
+ def resolve_output_path(
23
+ source,
24
+ fmt: str,
25
+ used_paths: set[Path],
26
+ preferred_filename: str | None = None,
27
+ ) -> Path:
28
+ """
29
+ Pick a repo-relative output path for one source, widening the parent prefix
30
+ until it no longer collides with a path already handed out.
31
+
32
+ Two sources sharing a stem (``alpha/report.pdf`` and ``beta/report.pdf``)
33
+ must not both resolve to ``report.md`` — the second would silently
34
+ overwrite the first.
35
+ """
36
+ source_path = Path(source)
37
+ filename = preferred_filename or output_filename(source_path, fmt)
38
+ parent_parts = [
39
+ part for part in source_path.parent.parts if part not in {"", ".", source_path.anchor}
40
+ ]
41
+
42
+ def with_prefix(width: int) -> Path:
43
+ return Path(*parent_parts[-width:]) / filename if width else Path(filename)
44
+
45
+ if source_path.is_absolute():
46
+ widths: Iterable[int] = range(0, len(parent_parts) + 1)
47
+ else:
48
+ widths = [len(parent_parts), *range(1, len(parent_parts)), 0]
49
+
50
+ for width in widths:
51
+ candidate = with_prefix(width)
52
+ if candidate not in used_paths:
53
+ used_paths.add(candidate)
54
+ return candidate
55
+
56
+ stem, suffix = Path(filename).stem, Path(filename).suffix
57
+ counter = 1
58
+ while True:
59
+ candidate = Path(f"{stem}-{counter}{suffix}")
60
+ if candidate not in used_paths:
61
+ used_paths.add(candidate)
62
+ return candidate
63
+ counter += 1
64
+
65
+
66
+ def resolve_output_paths(sources: Iterable, fmt: str) -> list[Path]:
67
+ """
68
+ Resolve every source to a distinct relative output path, in input order.
69
+
70
+ Collisions get the disambiguator that actually carries information:
71
+ same-stem siblings in one directory (``report.pdf`` next to ``report.docx``)
72
+ are told apart by source format, because widening the parent prefix would
73
+ only scatter siblings across unrelated output directories. Same-stem sources
74
+ in *different* directories keep the plain name and widen the prefix instead.
75
+ """
76
+ sources = list(sources)
77
+ sibling_counts = Counter((Path(source).parent, Path(source).stem) for source in sources)
78
+
79
+ used_paths: set[Path] = set()
80
+ resolved: list[Path] = []
81
+ for source in sources:
82
+ source_path = Path(source)
83
+ has_same_dir_twin = sibling_counts[(source_path.parent, source_path.stem)] > 1
84
+ preferred = extension_tagged_filename(source, fmt) if has_same_dir_twin else None
85
+ resolved.append(resolve_output_path(source, fmt, used_paths, preferred))
86
+ return resolved
@@ -0,0 +1,523 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from collections.abc import Iterable, Iterator
5
+ from dataclasses import asdict
6
+ from pathlib import Path
7
+
8
+ from langparse.chunkers.profiles import (
9
+ ChunkProfileNotSupportedError,
10
+ resolve_workbook_chunk_policy,
11
+ )
12
+ from langparse.chunkers.registry import create_chunker
13
+ from langparse.config import settings
14
+ from langparse.core.rendering import document_from_result
15
+ from langparse.engines.pdf.deepdoc_engine import DeepDocEngine
16
+ from langparse.engines.pdf.mineru import MinerUEngine
17
+ from langparse.engines.pdf.other import PaddleOCRVLEngine
18
+ from langparse.engines.pdf.simple import SimplePDFEngine
19
+ from langparse.engines.pdf.vision_llm import VisionLLMEngine
20
+ from langparse.parsers.registry import (
21
+ is_supported,
22
+ parser_kind_for,
23
+ unsupported_extension_error,
24
+ )
25
+ from langparse.progress import ProgressCallback, ProgressReporter
26
+ from langparse.services.output_paths import (
27
+ output_filename,
28
+ resolve_output_path,
29
+ resolve_output_paths,
30
+ )
31
+ from langparse.types import (
32
+ Chunk,
33
+ Document,
34
+ ParsedDocumentResult,
35
+ ParseDiagnostics,
36
+ ParsedPageResult,
37
+ )
38
+ from langparse.workbooks.modeling import WorkbookDisambiguation
39
+ from langparse.workbooks.types import WorkbookIR
40
+
41
+ #: Engines a caller can actually select. Advertising an engine that raises
42
+ #: NotImplementedError only once it runs wastes the user's configuration effort
43
+ #: and, for MinerU-scale setups, their model downloads.
44
+ ENGINE_MAP = {
45
+ "simple": SimplePDFEngine,
46
+ "mineru": MinerUEngine,
47
+ "deepdoc": DeepDocEngine,
48
+ }
49
+
50
+ #: Reserved names with adapters in the tree but no working implementation.
51
+ #: Selecting one fails immediately with an explanation instead of at parse time.
52
+ PLANNED_ENGINES = {
53
+ "vision_llm": VisionLLMEngine,
54
+ "paddle": PaddleOCRVLEngine,
55
+ }
56
+
57
+
58
+ class ParseService:
59
+ def chunk_result(
60
+ self,
61
+ parsed: ParsedDocumentResult,
62
+ chunker=None,
63
+ *,
64
+ chunk_profile: str | None = None,
65
+ chunk_strategy: str | None = None,
66
+ chunk_options: dict | None = None,
67
+ ) -> list[Chunk]:
68
+ """Chunk a parse result from its richest available representation."""
69
+ if chunker is not None and (chunk_strategy is not None or chunk_options is not None):
70
+ raise ValueError("custom chunker and chunk_strategy/options are mutually exclusive")
71
+ if chunker is not None and chunk_profile is not None:
72
+ raise ValueError("custom chunker and chunk_profile are mutually exclusive")
73
+
74
+ if chunker is not None:
75
+ if parsed.structure is not None and parsed.structure.kind == "workbook":
76
+ return chunker.chunk(parsed)
77
+ return chunker.chunk(document_from_result(parsed))
78
+
79
+ selected = create_chunker(chunk_strategy or "semantic", **(chunk_options or {}))
80
+ policy = resolve_workbook_chunk_policy(chunk_profile)
81
+ if isinstance(parsed.structure, WorkbookIR):
82
+ from langparse.chunkers.workbook import WorkbookStructuralChunker
83
+
84
+ return WorkbookStructuralChunker(profile=policy.name).chunk(parsed)
85
+
86
+ if policy.name.value == "analysis":
87
+ raise ChunkProfileNotSupportedError("analysis chunk profile requires WorkbookIR")
88
+
89
+ chunks = selected.chunk(document_from_result(parsed))
90
+ for chunk in chunks:
91
+ chunk.metadata["chunk_strategy"] = chunk_strategy or "semantic"
92
+ chunk.metadata["chunk_profile"] = policy.name.value
93
+ chunk.metadata["chunk_profile_version"] = policy.version
94
+ return chunks
95
+
96
+ def render_output(
97
+ self,
98
+ parsed: ParsedDocumentResult,
99
+ fmt: str,
100
+ chunks: list[Chunk] | None = None,
101
+ ) -> str:
102
+ if fmt == "markdown":
103
+ if chunks is None:
104
+ return parsed.markdown_content
105
+ if not chunks:
106
+ return parsed.markdown_content
107
+ return "\n\n---\n\n".join(chunk.content for chunk in chunks)
108
+ if fmt == "workbook-json":
109
+ from langparse.workbooks.bundle import WorkbookBundle
110
+
111
+ return WorkbookBundle.from_result(parsed).to_json()
112
+ if fmt == "json":
113
+ payload = asdict(parsed)
114
+ if chunks is not None:
115
+ payload["chunks"] = [asdict(chunk) for chunk in chunks]
116
+ return json.dumps(payload, ensure_ascii=False, indent=2, default=_json_scalar)
117
+ raise ValueError(f"Unsupported output format: {fmt}")
118
+
119
+ def parse_output(
120
+ self,
121
+ file_path,
122
+ engine_name="simple",
123
+ fmt="markdown",
124
+ engine=None,
125
+ chunk=False,
126
+ chunk_profile: str | None = None,
127
+ chunk_strategy: str | None = None,
128
+ chunk_options: dict | None = None,
129
+ workbook_disambiguation: WorkbookDisambiguation | None = None,
130
+ **kwargs,
131
+ ) -> str:
132
+ parsed = self.parse_result(
133
+ file_path,
134
+ engine_name=engine_name,
135
+ engine=engine,
136
+ chunk=chunk,
137
+ chunk_profile=chunk_profile,
138
+ chunk_strategy=chunk_strategy,
139
+ chunk_options=chunk_options,
140
+ workbook_disambiguation=workbook_disambiguation,
141
+ **kwargs,
142
+ )
143
+ return self.render_output(parsed, fmt, chunks=parsed.chunks if chunk else None)
144
+
145
+ def parse_batch_outputs(
146
+ self,
147
+ inputs,
148
+ engine_name="simple",
149
+ fmt="markdown",
150
+ engine=None,
151
+ chunk=False,
152
+ chunk_profile: str | None = None,
153
+ chunk_strategy: str | None = None,
154
+ chunk_options: dict | None = None,
155
+ workbook_disambiguation: WorkbookDisambiguation | None = None,
156
+ progress_callback: ProgressCallback | None = None,
157
+ **kwargs,
158
+ ) -> list[tuple[Path, str]]:
159
+ # `chunk` is named explicitly rather than left in **kwargs: kwargs also
160
+ # feed engine construction, and MinerU folds unknown kwargs into
161
+ # extra_options and sends them to its API as form fields.
162
+ model = kwargs.pop("model", None)
163
+ api_key = kwargs.pop("api_key", None)
164
+ base_url = kwargs.pop("base_url", None)
165
+ kwargs.pop("disambiguation", None)
166
+ outputs = []
167
+ active_engine = engine or self._create_engine(engine_name, **kwargs)
168
+ for file_path in self.expand_inputs(inputs):
169
+ outputs.append(
170
+ (
171
+ file_path,
172
+ self.parse_output(
173
+ file_path,
174
+ engine_name=engine_name,
175
+ fmt=fmt,
176
+ engine=active_engine,
177
+ chunk=chunk,
178
+ chunk_profile=chunk_profile,
179
+ chunk_strategy=chunk_strategy,
180
+ chunk_options=chunk_options,
181
+ workbook_disambiguation=workbook_disambiguation,
182
+ **(
183
+ {"progress_callback": progress_callback}
184
+ if progress_callback is not None
185
+ else {}
186
+ ),
187
+ model=model,
188
+ api_key=api_key,
189
+ base_url=base_url,
190
+ **kwargs,
191
+ ),
192
+ )
193
+ )
194
+ return outputs
195
+
196
+ def write_output(self, content: str, output_path) -> Path:
197
+ destination = Path(output_path)
198
+ destination.parent.mkdir(parents=True, exist_ok=True)
199
+ destination.write_text(content, encoding="utf-8")
200
+ return destination
201
+
202
+ def write_batch_outputs(self, outputs, output_dir, fmt: str) -> list[Path]:
203
+ output_dir = Path(output_dir)
204
+ output_dir.mkdir(parents=True, exist_ok=True)
205
+
206
+ outputs = list(outputs)
207
+ # Resolve all destinations together so same-stem siblings are grouped
208
+ # the same way BatchParseService groups them.
209
+ relative_paths = resolve_output_paths([source for source, _ in outputs], fmt)
210
+
211
+ written_paths = []
212
+ for (_, content), relative in zip(outputs, relative_paths, strict=True):
213
+ destination = output_dir / relative
214
+ self.write_output(content, destination)
215
+ written_paths.append(destination)
216
+ return written_paths
217
+
218
+ def expand_inputs(self, inputs):
219
+ paths = []
220
+ for item in self._flatten_inputs(inputs):
221
+ path = Path(item)
222
+ if not path.exists():
223
+ raise FileNotFoundError(f"File not found: {path}")
224
+ if path.is_dir():
225
+ paths.extend(
226
+ sorted(
227
+ child for child in path.iterdir() if child.is_file() and is_supported(child)
228
+ )
229
+ )
230
+ else:
231
+ paths.append(path)
232
+ return paths
233
+
234
+ def parse_result(
235
+ self,
236
+ file_path,
237
+ engine_name="simple",
238
+ engine=None,
239
+ chunk=False,
240
+ chunk_profile: str | None = None,
241
+ chunk_strategy: str | None = None,
242
+ chunk_options: dict | None = None,
243
+ workbook_disambiguation: str | WorkbookDisambiguation | None = None,
244
+ model: str | None = None,
245
+ api_key: str | None = None,
246
+ base_url: str | None = None,
247
+ progress_callback: ProgressCallback | None = None,
248
+ **kwargs,
249
+ ):
250
+ """
251
+ Parse any supported format into a ParsedDocumentResult.
252
+
253
+ This is the one place extension routing happens; everything else reads
254
+ the mapping from `langparse.parsers.registry`.
255
+ """
256
+ path = Path(file_path)
257
+ reporter = ProgressReporter(path, progress_callback)
258
+ with reporter.operation("file"):
259
+ if chunk:
260
+ resolve_workbook_chunk_policy(chunk_profile)
261
+ create_chunker(chunk_strategy or "semantic", **(chunk_options or {}))
262
+ if not path.exists():
263
+ raise FileNotFoundError(f"File not found: {path}")
264
+ kind = parser_kind_for(path)
265
+ if kind is None:
266
+ raise unsupported_extension_error(path)
267
+ reporter.emit("parsing")
268
+ if progress_callback is not None:
269
+ kwargs["progress_callback"] = progress_callback
270
+ if kind == "pdf":
271
+ parsed = self._collect_pdf_document_result(
272
+ path,
273
+ engine_name=engine_name,
274
+ engine=engine,
275
+ **kwargs,
276
+ )
277
+ else:
278
+ parsed = self._parser_for_kind(
279
+ kind,
280
+ workbook_disambiguation,
281
+ model=model,
282
+ api_key=api_key,
283
+ base_url=base_url,
284
+ ).parse_result(path, **kwargs)
285
+ if chunk:
286
+ reporter.emit("chunking")
287
+ self._populate_chunks(parsed, chunk_profile, chunk_strategy, chunk_options)
288
+ return parsed
289
+
290
+ def _populate_chunks(
291
+ self,
292
+ parsed: ParsedDocumentResult,
293
+ chunk_profile: str | None,
294
+ chunk_strategy: str | None = None,
295
+ chunk_options: dict | None = None,
296
+ ) -> None:
297
+ policy = resolve_workbook_chunk_policy(chunk_profile)
298
+ try:
299
+ parsed.chunks = self.chunk_result(
300
+ parsed,
301
+ chunk_profile=policy.name.value,
302
+ chunk_strategy=chunk_strategy,
303
+ chunk_options=chunk_options,
304
+ )
305
+ except ChunkProfileNotSupportedError:
306
+ if parsed.diagnostics is None:
307
+ parsed.diagnostics = ParseDiagnostics()
308
+ if parsed.diagnostics.status != "failed":
309
+ parsed.diagnostics.status = "partial"
310
+ parsed.diagnostics.unsupported_features.append(
311
+ f"Chunking profile '{policy.name.value}' is not supported for engine "
312
+ f"'{parsed.engine}'."
313
+ )
314
+ parsed.chunks = []
315
+ except Exception as exc: # noqa: BLE001 - preserve parsed result at chunk boundary
316
+ if parsed.diagnostics is None:
317
+ parsed.diagnostics = ParseDiagnostics()
318
+ if parsed.diagnostics.status != "failed":
319
+ parsed.diagnostics.status = "partial"
320
+ parsed.diagnostics.errors.append(
321
+ f"Chunking profile '{policy.name.value}' failed ({type(exc).__name__})."
322
+ )
323
+ parsed.chunks = []
324
+
325
+ def _parser_for_kind(
326
+ self,
327
+ kind: str,
328
+ workbook_disambiguation: str | WorkbookDisambiguation | None = None,
329
+ *,
330
+ model: str | None = None,
331
+ api_key: str | None = None,
332
+ base_url: str | None = None,
333
+ ):
334
+ if kind == "docx":
335
+ from langparse.parsers.docx_parser import DocxParser
336
+
337
+ return DocxParser()
338
+ if kind == "excel":
339
+ from langparse.parsers.excel_parser import ExcelParser
340
+
341
+ return ExcelParser(
342
+ disambiguation=workbook_disambiguation,
343
+ model=model,
344
+ api_key=api_key,
345
+ base_url=base_url,
346
+ )
347
+ if kind == "markdown":
348
+ from langparse.parsers.markdown_parser import MarkdownParser
349
+
350
+ return MarkdownParser()
351
+ raise ValueError(f"No parser registered for kind: {kind}")
352
+
353
+ def parse_file(self, file_path, engine_name="simple", engine=None, **kwargs):
354
+ parsed = self.parse_result(
355
+ file_path,
356
+ engine_name=engine_name,
357
+ engine=engine,
358
+ **kwargs,
359
+ )
360
+ return self._build_document_from_result(parsed)
361
+
362
+ def parse_pdf_document(self, file_path, engine_name="simple", engine=None, **kwargs):
363
+ return self.parse_file(file_path, engine_name=engine_name, engine=engine, **kwargs)
364
+
365
+ def parse_batch(self, inputs, engine_name="simple", engine=None, **kwargs):
366
+ progress_callback = kwargs.pop("progress_callback", None)
367
+ workbook_disambiguation = kwargs.pop("workbook_disambiguation", None)
368
+ model = kwargs.pop("model", None)
369
+ api_key = kwargs.pop("api_key", None)
370
+ base_url = kwargs.pop("base_url", None)
371
+ chunk_kwargs = {
372
+ key: kwargs.pop(key)
373
+ for key in ("chunk", "chunk_profile", "chunk_strategy", "chunk_options")
374
+ if key in kwargs
375
+ }
376
+ documents = []
377
+ active_engine = engine or self._create_engine(engine_name, **kwargs)
378
+ for file_path in self.expand_inputs(inputs):
379
+ documents.append(
380
+ self.parse_file(
381
+ file_path,
382
+ engine_name=engine_name,
383
+ engine=active_engine,
384
+ workbook_disambiguation=workbook_disambiguation,
385
+ **(
386
+ {"progress_callback": progress_callback}
387
+ if progress_callback is not None
388
+ else {}
389
+ ),
390
+ model=model,
391
+ api_key=api_key,
392
+ base_url=base_url,
393
+ **chunk_kwargs,
394
+ **kwargs,
395
+ )
396
+ )
397
+ return documents
398
+
399
+ def _collect_pdf_document_result(
400
+ self, file_path, engine_name="simple", engine=None, progress_callback=None, **kwargs
401
+ ):
402
+ file_path = Path(file_path)
403
+ if not file_path.exists():
404
+ raise FileNotFoundError(f"File not found: {file_path}")
405
+
406
+ engine_kwargs = {
407
+ k: v
408
+ for k, v in kwargs.items()
409
+ if k
410
+ not in (
411
+ "model",
412
+ "api_key",
413
+ "base_url",
414
+ "workbook_disambiguation",
415
+ "disambiguation",
416
+ )
417
+ }
418
+ active_engine = engine or self._create_engine(engine_name, **engine_kwargs)
419
+ if progress_callback is not None:
420
+ kwargs["progress_callback"] = progress_callback
421
+ if hasattr(active_engine, "process_document"):
422
+ process_document = active_engine.process_document
423
+ if not callable(process_document):
424
+ raise TypeError(
425
+ f"{type(active_engine).__name__}.process_document exists but is not callable"
426
+ )
427
+
428
+ parsed = process_document(file_path, **kwargs)
429
+ if not isinstance(parsed, ParsedDocumentResult):
430
+ raise TypeError(
431
+ f"{type(active_engine).__name__}.process_document must return ParsedDocumentResult"
432
+ )
433
+ return parsed
434
+
435
+ pages = []
436
+ for page in active_engine.process(file_path, **kwargs):
437
+ pages.append(self._to_parsed_page_result(page))
438
+
439
+ return ParsedDocumentResult(
440
+ source=str(file_path),
441
+ filename=file_path.name,
442
+ engine=engine_name,
443
+ pages=pages,
444
+ markdown_content="\n".join(page.markdown_content for page in pages),
445
+ metadata=self._document_metadata_from_pages(pages),
446
+ )
447
+
448
+ def _document_metadata_from_pages(self, pages: list[ParsedPageResult]) -> dict:
449
+ """Roll per-page engine signals up to the document, where metrics read them."""
450
+ return {
451
+ "ocr_applied": any(page.metadata.get("ocr_applied") for page in pages),
452
+ "ocr_text_chars": sum(
453
+ int(page.metadata.get("ocr_text_chars", 0) or 0) for page in pages
454
+ ),
455
+ }
456
+
457
+ def create_engine(self, engine_name: str = "simple", **kwargs):
458
+ """
459
+ Build one engine instance callers can reuse across many files.
460
+
461
+ Batch runs must share a single engine: a per-file MinerU engine would
462
+ start and stop its own local mineru-api service, and concurrent workers
463
+ would race for the same port.
464
+ """
465
+ return self._create_engine(engine_name, **kwargs)
466
+
467
+ def _create_engine(self, engine_name: str, **kwargs):
468
+ engine_class = ENGINE_MAP.get(engine_name)
469
+ if engine_class is None:
470
+ available = ", ".join(sorted(ENGINE_MAP))
471
+ if engine_name in PLANNED_ENGINES:
472
+ raise ValueError(
473
+ f"Engine '{engine_name}' is not implemented yet. Available: {available}"
474
+ )
475
+ raise ValueError(f"Unknown engine: {engine_name}. Available: {available}")
476
+
477
+ engine_config = settings.resolve_engine_config(engine_name, kwargs)
478
+ return engine_class(**engine_config)
479
+
480
+ def _to_parsed_page_result(self, page) -> ParsedPageResult:
481
+ return ParsedPageResult(
482
+ page_number=page.page_number,
483
+ markdown_content=page.markdown_content,
484
+ plain_text=getattr(page, "plain_text", ""),
485
+ elements=list(getattr(page, "elements", [])),
486
+ tables=list(getattr(page, "tables", [])),
487
+ images=list(getattr(page, "images", [])),
488
+ metadata=dict(getattr(page, "metadata", {})),
489
+ )
490
+
491
+ def _build_document_from_result(self, parsed: ParsedDocumentResult) -> Document:
492
+ return document_from_result(parsed)
493
+
494
+ def _flatten_inputs(self, inputs) -> Iterator[str | Path]:
495
+ if isinstance(inputs, (str, Path)):
496
+ yield inputs
497
+ return
498
+
499
+ if isinstance(inputs, Iterable):
500
+ for item in inputs:
501
+ if isinstance(item, (str, Path)):
502
+ yield item
503
+ elif isinstance(item, Iterable):
504
+ yield from self._flatten_inputs(item)
505
+ else:
506
+ yield item
507
+ return
508
+
509
+ yield inputs
510
+
511
+ def _output_filename(self, source, fmt: str) -> str:
512
+ return output_filename(source, fmt)
513
+
514
+ def _output_path_for_batch_item(self, source, fmt: str, used_paths: set[Path]) -> Path:
515
+ return resolve_output_path(source, fmt, used_paths)
516
+
517
+
518
+ def _json_scalar(value):
519
+ """Serialize native spreadsheet scalars such as dates and decimals."""
520
+
521
+ if hasattr(value, "isoformat"):
522
+ return value.isoformat()
523
+ return str(value)
@@ -0,0 +1,65 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass, field
4
+
5
+ from langparse.metrics import ParseMetrics
6
+
7
+
8
+ @dataclass
9
+ class QualityCheck:
10
+ min_pages: int | None = None
11
+ min_chars: int | None = None
12
+ min_tables: int | None = None
13
+ min_images: int | None = None
14
+ require_page_markers: bool = False
15
+ require_table_markdown: bool = False
16
+ require_ocr_text: bool = False
17
+ require_multi_column_check: bool = False
18
+ max_header_footer_repetition_ratio: float | None = None
19
+ require_captions_for_images: bool = False
20
+
21
+
22
+ @dataclass
23
+ class QualityCheckResult:
24
+ passed: bool
25
+ failures: list[str] = field(default_factory=list)
26
+ warnings: list[str] = field(default_factory=list)
27
+
28
+
29
+ def run_quality_checks(metrics: ParseMetrics, checks: QualityCheck) -> QualityCheckResult:
30
+ failures: list[str] = []
31
+ warnings: list[str] = []
32
+
33
+ if checks.min_pages is not None and metrics.page_count < checks.min_pages:
34
+ failures.append("min_pages")
35
+ if checks.min_chars is not None and metrics.markdown_chars < checks.min_chars:
36
+ failures.append("min_chars")
37
+ if checks.min_tables is not None and metrics.table_count < checks.min_tables:
38
+ failures.append("min_tables")
39
+ if checks.min_images is not None and metrics.image_count < checks.min_images:
40
+ failures.append("min_images")
41
+ if checks.require_page_markers and metrics.page_marker_coverage <= 0:
42
+ failures.append("require_page_markers")
43
+ if checks.require_table_markdown and metrics.table_count <= 0:
44
+ failures.append("require_table_markdown")
45
+ if checks.require_ocr_text and metrics.ocr_text_chars <= 0:
46
+ failures.append("require_ocr_text")
47
+ if (
48
+ checks.require_multi_column_check
49
+ and not metrics.multi_column_detected
50
+ and metrics.reading_order_warnings == 0
51
+ ):
52
+ failures.append("require_multi_column_check")
53
+ if (
54
+ checks.require_captions_for_images
55
+ and metrics.image_count > 0
56
+ and metrics.images_with_caption_ratio < 1.0
57
+ ):
58
+ failures.append("require_captions_for_images")
59
+ if (
60
+ checks.max_header_footer_repetition_ratio is not None
61
+ and metrics.header_footer_removed_count == 0
62
+ ):
63
+ warnings.append("header_footer_filter_not_applied")
64
+
65
+ return QualityCheckResult(passed=not failures, failures=failures, warnings=warnings)