langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
import threading
|
|
2
|
+
from collections.abc import Iterator
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from langparse.core.engine import BaseEngine, PageResult
|
|
6
|
+
from langparse.engines.pdf.ocr import (
|
|
7
|
+
DEFAULT_MIN_CHARS,
|
|
8
|
+
DEFAULT_RESOLUTION,
|
|
9
|
+
load_recogniser,
|
|
10
|
+
needs_ocr,
|
|
11
|
+
ocr_page_text,
|
|
12
|
+
)
|
|
13
|
+
from langparse.progress import ProgressCallback, ProgressReporter
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class BasePDFEngine(BaseEngine):
|
|
17
|
+
"""
|
|
18
|
+
Base class specifically for PDF engines.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
def __init__(self, **kwargs):
|
|
22
|
+
pass
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class SimplePDFEngine(BasePDFEngine):
|
|
26
|
+
"""
|
|
27
|
+
A lightweight, dependency-free (except pdfplumber) engine.
|
|
28
|
+
Good for simple, native PDFs.
|
|
29
|
+
|
|
30
|
+
Scanned pages fall back to OCR. Without it a scanned document parses
|
|
31
|
+
"successfully" into a handful of watermark characters, which is harder for a
|
|
32
|
+
caller to notice than an outright failure.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
def __init__(
|
|
36
|
+
self,
|
|
37
|
+
enable_ocr: bool = True,
|
|
38
|
+
ocr_min_chars: int = DEFAULT_MIN_CHARS,
|
|
39
|
+
ocr_resolution: int = DEFAULT_RESOLUTION,
|
|
40
|
+
recogniser=None,
|
|
41
|
+
**kwargs,
|
|
42
|
+
):
|
|
43
|
+
self.enable_ocr = enable_ocr
|
|
44
|
+
self.ocr_min_chars = ocr_min_chars
|
|
45
|
+
self.ocr_resolution = ocr_resolution
|
|
46
|
+
self._recogniser = recogniser
|
|
47
|
+
# Batch runs share one engine across worker threads. The lock keeps
|
|
48
|
+
# model loading (tens of seconds) from happening once per thread, and
|
|
49
|
+
# serialises recognition because rapidocr states no thread-safety
|
|
50
|
+
# guarantee -- correctness over throughput on an already slow path.
|
|
51
|
+
self._ocr_lock = threading.Lock()
|
|
52
|
+
|
|
53
|
+
def process(
|
|
54
|
+
self, file_path: Path, *, progress_callback: ProgressCallback | None = None, **kwargs
|
|
55
|
+
) -> Iterator[PageResult]:
|
|
56
|
+
reporter = ProgressReporter(str(file_path), progress_callback)
|
|
57
|
+
try:
|
|
58
|
+
import pdfplumber
|
|
59
|
+
except ImportError:
|
|
60
|
+
raise ImportError("Please install `pdfplumber` to use the 'simple' engine.") from None
|
|
61
|
+
|
|
62
|
+
with pdfplumber.open(file_path) as pdf:
|
|
63
|
+
for i, page in enumerate(pdf.pages):
|
|
64
|
+
text = page.extract_text() or ""
|
|
65
|
+
ocr_applied = False
|
|
66
|
+
|
|
67
|
+
if self.enable_ocr and needs_ocr(page, min_chars=self.ocr_min_chars):
|
|
68
|
+
recovered = self._run_ocr(page)
|
|
69
|
+
if recovered:
|
|
70
|
+
text = recovered
|
|
71
|
+
ocr_applied = True
|
|
72
|
+
|
|
73
|
+
tables, table_markdown = self._extract_tables(page)
|
|
74
|
+
|
|
75
|
+
markdown_content = text
|
|
76
|
+
if table_markdown:
|
|
77
|
+
markdown_content = "\n\n".join([text, "\n".join(table_markdown)]).strip()
|
|
78
|
+
|
|
79
|
+
reporter.emit(
|
|
80
|
+
"parsing", completed_units=i + 1, total_units=len(pdf.pages), unit="pages"
|
|
81
|
+
)
|
|
82
|
+
yield PageResult(
|
|
83
|
+
page_number=i + 1,
|
|
84
|
+
markdown_content=markdown_content,
|
|
85
|
+
plain_text=text,
|
|
86
|
+
elements=[],
|
|
87
|
+
tables=tables,
|
|
88
|
+
images=[],
|
|
89
|
+
metadata={
|
|
90
|
+
"engine_name": "simple",
|
|
91
|
+
"ocr_applied": ocr_applied,
|
|
92
|
+
"ocr_text_chars": len(text) if ocr_applied else 0,
|
|
93
|
+
},
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
def _run_ocr(self, page) -> str:
|
|
97
|
+
"""Recognise a page, degrading to no text rather than failing the parse."""
|
|
98
|
+
try:
|
|
99
|
+
with self._ocr_lock:
|
|
100
|
+
recogniser = self._build_recogniser_if_needed()
|
|
101
|
+
return ocr_page_text(page, recogniser, resolution=self.ocr_resolution)
|
|
102
|
+
except ImportError:
|
|
103
|
+
# rapidocr absent: the page keeps whatever thin text layer it had.
|
|
104
|
+
return ""
|
|
105
|
+
|
|
106
|
+
def _resolve_recogniser(self):
|
|
107
|
+
with self._ocr_lock:
|
|
108
|
+
return self._build_recogniser_if_needed()
|
|
109
|
+
|
|
110
|
+
def _build_recogniser_if_needed(self):
|
|
111
|
+
"""Caller must hold `_ocr_lock`."""
|
|
112
|
+
if self._recogniser is None:
|
|
113
|
+
self._recogniser = load_recogniser()
|
|
114
|
+
return self._recogniser
|
|
115
|
+
|
|
116
|
+
def _extract_tables(self, page):
|
|
117
|
+
tables = []
|
|
118
|
+
table_markdown = []
|
|
119
|
+
extract_tables = getattr(page, "extract_tables", None)
|
|
120
|
+
for table in (extract_tables() if callable(extract_tables) else []) or []:
|
|
121
|
+
cleaned_table = [
|
|
122
|
+
["" if cell is None else str(cell).strip().replace("\n", " ") for cell in row]
|
|
123
|
+
for row in table
|
|
124
|
+
]
|
|
125
|
+
if not cleaned_table:
|
|
126
|
+
continue
|
|
127
|
+
|
|
128
|
+
tables.append({"rows": cleaned_table})
|
|
129
|
+
headers = cleaned_table[0]
|
|
130
|
+
table_markdown.append(f"| {' | '.join(headers)} |")
|
|
131
|
+
table_markdown.append(f"| {' | '.join(['---'] * len(headers))} |")
|
|
132
|
+
for row in cleaned_table[1:]:
|
|
133
|
+
table_markdown.append(f"| {' | '.join(row)} |")
|
|
134
|
+
return tables, table_markdown
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
from collections.abc import Iterator
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
from langparse.core.engine import PageResult
|
|
5
|
+
from langparse.engines.pdf.simple import BasePDFEngine
|
|
6
|
+
from langparse.logging import get_logger
|
|
7
|
+
|
|
8
|
+
logger = get_logger(__name__)
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class VisionLLMEngine(BasePDFEngine):
|
|
12
|
+
"""
|
|
13
|
+
Uses Vision LLMs (GPT-4o, Gemini 1.5 Pro) to parse pages.
|
|
14
|
+
Extremely slow/expensive but handles EVERYTHING (handwriting, complex charts).
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
def __init__(self, model_name: str = "gpt-4o", api_key: str = None):
|
|
18
|
+
self.model_name = model_name
|
|
19
|
+
self.api_key = api_key
|
|
20
|
+
|
|
21
|
+
def process(self, file_path: Path, **kwargs) -> Iterator[PageResult]:
|
|
22
|
+
# 1. Convert PDF to Images (using pdf2image)
|
|
23
|
+
# 2. Send each image to LLM with a prompt like "Transcribe this page to Markdown"
|
|
24
|
+
|
|
25
|
+
logger.debug("VisionLLM processing %s with %s", file_path, self.model_name)
|
|
26
|
+
|
|
27
|
+
raise NotImplementedError("Vision LLM integration is pending.")
|
langparse/errors.py
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
from enum import Enum
|
|
6
|
+
from subprocess import TimeoutExpired
|
|
7
|
+
from urllib.error import URLError
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ErrorType(str, Enum):
|
|
11
|
+
DEPENDENCY_MISSING = "dependency_missing"
|
|
12
|
+
FILE_NOT_FOUND = "file_not_found"
|
|
13
|
+
UNSUPPORTED_FORMAT = "unsupported_format"
|
|
14
|
+
ENGINE_UNAVAILABLE = "engine_unavailable"
|
|
15
|
+
ENGINE_TIMEOUT = "engine_timeout"
|
|
16
|
+
PARSE_FAILED = "parse_failed"
|
|
17
|
+
QUALITY_CHECK_FAILED = "quality_check_failed"
|
|
18
|
+
OCR_UNAVAILABLE = "ocr_unavailable"
|
|
19
|
+
LAYOUT_QUALITY_WARNING = "layout_quality_warning"
|
|
20
|
+
TABLE_EXTRACTION_FAILED = "table_extraction_failed"
|
|
21
|
+
WORKBOOK_EVALUATION_FAILED = "workbook_evaluation_failed"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class ClassifiedError:
|
|
26
|
+
error_type: ErrorType
|
|
27
|
+
message: str
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def classify_exception(exc: BaseException) -> ClassifiedError:
|
|
31
|
+
message = str(exc)
|
|
32
|
+
lowered = message.lower()
|
|
33
|
+
|
|
34
|
+
from langparse.workbooks.evaluation.schema import WorkbookEvaluationError
|
|
35
|
+
|
|
36
|
+
if isinstance(exc, WorkbookEvaluationError):
|
|
37
|
+
return ClassifiedError(ErrorType.WORKBOOK_EVALUATION_FAILED, message)
|
|
38
|
+
if isinstance(exc, FileNotFoundError):
|
|
39
|
+
return ClassifiedError(ErrorType.FILE_NOT_FOUND, message)
|
|
40
|
+
if isinstance(exc, ImportError):
|
|
41
|
+
return ClassifiedError(ErrorType.DEPENDENCY_MISSING, message)
|
|
42
|
+
if "unsupported file extension" in lowered:
|
|
43
|
+
return ClassifiedError(ErrorType.UNSUPPORTED_FORMAT, message)
|
|
44
|
+
if "cuda" in lowered and "not available" in lowered:
|
|
45
|
+
return ClassifiedError(ErrorType.ENGINE_UNAVAILABLE, message)
|
|
46
|
+
if "unable to start local mineru-api" in lowered:
|
|
47
|
+
return ClassifiedError(ErrorType.ENGINE_UNAVAILABLE, message)
|
|
48
|
+
if _is_timeout(exc) or re.search(r"\btimed out\b", lowered):
|
|
49
|
+
return ClassifiedError(ErrorType.ENGINE_TIMEOUT, message)
|
|
50
|
+
if "ocr" in lowered and "unavailable" in lowered:
|
|
51
|
+
return ClassifiedError(ErrorType.OCR_UNAVAILABLE, message)
|
|
52
|
+
if "table" in lowered and "failed" in lowered:
|
|
53
|
+
return ClassifiedError(ErrorType.TABLE_EXTRACTION_FAILED, message)
|
|
54
|
+
|
|
55
|
+
return ClassifiedError(ErrorType.PARSE_FAILED, message)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _is_timeout(exc: BaseException) -> bool:
|
|
59
|
+
"""Recognize standard timeout types through explicit backend wrappers."""
|
|
60
|
+
seen: set[int] = set()
|
|
61
|
+
current: BaseException | None = exc
|
|
62
|
+
while current is not None and id(current) not in seen:
|
|
63
|
+
seen.add(id(current))
|
|
64
|
+
if isinstance(current, (TimeoutError, TimeoutExpired)):
|
|
65
|
+
return True
|
|
66
|
+
if isinstance(current, URLError) and isinstance(current.reason, BaseException):
|
|
67
|
+
current = current.reason
|
|
68
|
+
else:
|
|
69
|
+
current = current.__cause__
|
|
70
|
+
return False
|
langparse/logging.py
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Logging for the library.
|
|
3
|
+
|
|
4
|
+
A library should not choose a logging framework for the application embedding
|
|
5
|
+
it, nor write to stderr uninvited. The stdlib logger with a NullHandler stays
|
|
6
|
+
silent until the host configures logging, and costs no dependency -- which
|
|
7
|
+
leaves langparse installable with nothing but the extras a caller actually
|
|
8
|
+
needs for their formats.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import logging
|
|
14
|
+
|
|
15
|
+
ROOT_LOGGER_NAME = "langparse"
|
|
16
|
+
|
|
17
|
+
_root = logging.getLogger(ROOT_LOGGER_NAME)
|
|
18
|
+
_root.addHandler(logging.NullHandler())
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def get_logger(name: str | None = None) -> logging.Logger:
|
|
22
|
+
"""Return a logger under the `langparse` namespace."""
|
|
23
|
+
if not name:
|
|
24
|
+
return _root
|
|
25
|
+
if name.startswith(f"{ROOT_LOGGER_NAME}."):
|
|
26
|
+
return logging.getLogger(name)
|
|
27
|
+
return logging.getLogger(f"{ROOT_LOGGER_NAME}.{name}")
|
langparse/metrics.py
ADDED
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from dataclasses import dataclass, field
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from langparse.types import Chunk, ParsedDocumentResult
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def pages_per_second(page_count: int, elapsed_seconds: float) -> float:
|
|
11
|
+
if elapsed_seconds <= 0:
|
|
12
|
+
return 0.0
|
|
13
|
+
return round(page_count / elapsed_seconds, 4)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def count_markdown_tables(markdown: str) -> int:
|
|
17
|
+
separator_pattern = re.compile(r"^\s*\|?\s*:?-{3,}:?\s*(\|\s*:?-{3,}:?\s*)+\|?\s*$")
|
|
18
|
+
return sum(1 for line in markdown.splitlines() if separator_pattern.match(line))
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class ParseMetrics:
|
|
23
|
+
elapsed_seconds: float = 0.0
|
|
24
|
+
page_count: int = 0
|
|
25
|
+
pages_per_second: float = 0.0
|
|
26
|
+
output_bytes: int = 0
|
|
27
|
+
markdown_chars: int = 0
|
|
28
|
+
table_count: int = 0
|
|
29
|
+
image_count: int = 0
|
|
30
|
+
chunk_count: int = 0
|
|
31
|
+
chunks_with_page_numbers_ratio: float = 0.0
|
|
32
|
+
page_marker_coverage: float = 0.0
|
|
33
|
+
ocr_applied: bool = False
|
|
34
|
+
ocr_text_chars: int = 0
|
|
35
|
+
multi_column_detected: bool = False
|
|
36
|
+
reading_order_warnings: int = 0
|
|
37
|
+
header_footer_removed_count: int = 0
|
|
38
|
+
caption_count: int = 0
|
|
39
|
+
images_with_caption_ratio: float = 0.0
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class BatchItemResult:
|
|
44
|
+
source: str
|
|
45
|
+
status: str
|
|
46
|
+
output_path: str | None = None
|
|
47
|
+
metrics: ParseMetrics | None = None
|
|
48
|
+
error_type: str | None = None
|
|
49
|
+
error_message: str | None = None
|
|
50
|
+
engine: str | None = None
|
|
51
|
+
started_at: str | None = None
|
|
52
|
+
finished_at: str | None = None
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclass
|
|
56
|
+
class BatchRunResult:
|
|
57
|
+
items: list[BatchItemResult] = field(default_factory=list)
|
|
58
|
+
summary: dict[str, Any] = field(default_factory=dict)
|
|
59
|
+
#: Rendered documents, populated only when the run had no output directory
|
|
60
|
+
#: and therefore rendered to memory instead of to disk. Kept off
|
|
61
|
+
#: BatchItemResult so it never bloats the JSONL report.
|
|
62
|
+
rendered_outputs: list[str] = field(default_factory=list)
|
|
63
|
+
|
|
64
|
+
@property
|
|
65
|
+
def total_files(self) -> int:
|
|
66
|
+
return len(self.items)
|
|
67
|
+
|
|
68
|
+
@property
|
|
69
|
+
def success_count(self) -> int:
|
|
70
|
+
return sum(1 for item in self.items if item.status == "success")
|
|
71
|
+
|
|
72
|
+
@property
|
|
73
|
+
def failed_count(self) -> int:
|
|
74
|
+
return sum(1 for item in self.items if item.status == "failed")
|
|
75
|
+
|
|
76
|
+
@property
|
|
77
|
+
def skipped_count(self) -> int:
|
|
78
|
+
return sum(1 for item in self.items if item.status == "skipped")
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def page_marker_coverage(parsed: ParsedDocumentResult) -> float:
|
|
82
|
+
"""
|
|
83
|
+
Fraction of pages carrying a usable page number.
|
|
84
|
+
|
|
85
|
+
This is what makes a chunk citable back to a page, so the
|
|
86
|
+
``require_page_markers`` quality check reads it directly.
|
|
87
|
+
"""
|
|
88
|
+
if not getattr(parsed, "paginated", True) or not parsed.pages:
|
|
89
|
+
return 0.0
|
|
90
|
+
numbered = sum(1 for page in parsed.pages if page.page_number and page.page_number >= 1)
|
|
91
|
+
return round(numbered / len(parsed.pages), 4)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def collect_parse_metrics(
|
|
95
|
+
parsed: ParsedDocumentResult,
|
|
96
|
+
elapsed_seconds: float,
|
|
97
|
+
chunks: list[Chunk] | None = None,
|
|
98
|
+
) -> ParseMetrics:
|
|
99
|
+
markdown = parsed.markdown_content or ""
|
|
100
|
+
page_count = len(parsed.pages)
|
|
101
|
+
chunk_count = len(chunks) if chunks is not None else 0
|
|
102
|
+
chunks_with_pages = (
|
|
103
|
+
sum(1 for chunk in chunks if chunk.metadata.get("page_numbers")) if chunks else 0
|
|
104
|
+
)
|
|
105
|
+
image_count = sum(len(page.images) for page in parsed.pages)
|
|
106
|
+
table_count = sum(len(page.tables) for page in parsed.pages) or count_markdown_tables(markdown)
|
|
107
|
+
caption_count = sum(1 for page in parsed.pages for image in page.images if image.get("caption"))
|
|
108
|
+
|
|
109
|
+
return ParseMetrics(
|
|
110
|
+
elapsed_seconds=round(elapsed_seconds, 4),
|
|
111
|
+
page_count=page_count,
|
|
112
|
+
pages_per_second=pages_per_second(page_count, elapsed_seconds),
|
|
113
|
+
output_bytes=len(markdown.encode("utf-8")),
|
|
114
|
+
markdown_chars=len(markdown),
|
|
115
|
+
table_count=table_count,
|
|
116
|
+
image_count=image_count,
|
|
117
|
+
chunk_count=chunk_count,
|
|
118
|
+
chunks_with_page_numbers_ratio=round(chunks_with_pages / chunk_count, 4)
|
|
119
|
+
if chunk_count
|
|
120
|
+
else 0.0,
|
|
121
|
+
page_marker_coverage=page_marker_coverage(parsed),
|
|
122
|
+
ocr_applied=bool(parsed.metadata.get("ocr_applied", False)),
|
|
123
|
+
ocr_text_chars=int(parsed.metadata.get("ocr_text_chars", 0) or 0),
|
|
124
|
+
multi_column_detected=bool(parsed.metadata.get("multi_column_detected", False)),
|
|
125
|
+
reading_order_warnings=int(parsed.metadata.get("reading_order_warnings", 0) or 0),
|
|
126
|
+
header_footer_removed_count=int(parsed.metadata.get("header_footer_removed_count", 0) or 0),
|
|
127
|
+
caption_count=caption_count,
|
|
128
|
+
images_with_caption_ratio=round(caption_count / image_count, 4) if image_count else 0.0,
|
|
129
|
+
)
|
|
File without changes
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
|
|
3
|
+
from langparse.core.parser import BaseParser
|
|
4
|
+
from langparse.types import ParsedDocumentResult, ParsedElement, ParsedPageResult
|
|
5
|
+
|
|
6
|
+
_HEADING_PREFIXES = (
|
|
7
|
+
("heading 1", "# "),
|
|
8
|
+
("title", "# "),
|
|
9
|
+
("heading 2", "## "),
|
|
10
|
+
("heading 3", "### "),
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class DocxParser(BaseParser):
|
|
15
|
+
"""
|
|
16
|
+
Parses .docx files to Markdown.
|
|
17
|
+
Note: DOCX is a flow format, so 'page numbers' are not strictly defined.
|
|
18
|
+
We treat the entire document as Page 1 for now, unless we convert to PDF first.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
def parse_result(self, file_path: str | Path, **kwargs) -> ParsedDocumentResult:
|
|
22
|
+
path = self._resolve_existing_path(file_path)
|
|
23
|
+
|
|
24
|
+
try:
|
|
25
|
+
import docx
|
|
26
|
+
except ImportError:
|
|
27
|
+
raise ImportError(
|
|
28
|
+
"python-docx is required. Install with `pip install python-docx`."
|
|
29
|
+
) from None
|
|
30
|
+
|
|
31
|
+
doc = docx.Document(path)
|
|
32
|
+
markdown_lines: list[str] = []
|
|
33
|
+
elements: list[ParsedElement] = []
|
|
34
|
+
tables: list[dict] = []
|
|
35
|
+
|
|
36
|
+
for block in self._iter_block_items(docx, doc):
|
|
37
|
+
if isinstance(block, docx.text.paragraph.Paragraph):
|
|
38
|
+
rendered = self._render_paragraph(block)
|
|
39
|
+
if rendered is None:
|
|
40
|
+
continue
|
|
41
|
+
kind, markdown = rendered
|
|
42
|
+
markdown_lines.extend([markdown, ""])
|
|
43
|
+
elements.append(ParsedElement(kind=kind, text=block.text.strip()))
|
|
44
|
+
continue
|
|
45
|
+
|
|
46
|
+
rows = self._table_rows(block)
|
|
47
|
+
if not rows:
|
|
48
|
+
continue
|
|
49
|
+
table_markdown = self._rows_to_markdown(rows)
|
|
50
|
+
tables.append({"rows": rows})
|
|
51
|
+
markdown_lines.extend([table_markdown, ""])
|
|
52
|
+
elements.append(ParsedElement(kind="table", text=table_markdown))
|
|
53
|
+
|
|
54
|
+
markdown_content = "\n".join(markdown_lines)
|
|
55
|
+
return ParsedDocumentResult(
|
|
56
|
+
source=str(path),
|
|
57
|
+
filename=path.name,
|
|
58
|
+
engine="docx",
|
|
59
|
+
pages=[
|
|
60
|
+
ParsedPageResult(
|
|
61
|
+
page_number=1,
|
|
62
|
+
markdown_content=markdown_content,
|
|
63
|
+
plain_text="\n".join(
|
|
64
|
+
element.text for element in elements if element.kind != "table"
|
|
65
|
+
),
|
|
66
|
+
elements=elements,
|
|
67
|
+
tables=tables,
|
|
68
|
+
)
|
|
69
|
+
],
|
|
70
|
+
markdown_content=markdown_content,
|
|
71
|
+
metadata={"extension": ".docx"},
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
def _iter_block_items(self, docx, parent):
|
|
75
|
+
"""
|
|
76
|
+
Yield paragraphs and tables in document order.
|
|
77
|
+
|
|
78
|
+
python-docx exposes paragraphs and tables as separate collections, which
|
|
79
|
+
loses their interleaving, so walk the body XML directly.
|
|
80
|
+
"""
|
|
81
|
+
from docx.oxml.table import CT_Tbl
|
|
82
|
+
from docx.oxml.text.paragraph import CT_P
|
|
83
|
+
|
|
84
|
+
parent_elm = (
|
|
85
|
+
parent.element.body if isinstance(parent, docx.document.Document) else parent._element
|
|
86
|
+
)
|
|
87
|
+
for child in parent_elm.iterchildren():
|
|
88
|
+
if isinstance(child, CT_P):
|
|
89
|
+
yield docx.text.paragraph.Paragraph(child, parent)
|
|
90
|
+
elif isinstance(child, CT_Tbl):
|
|
91
|
+
yield docx.table.Table(child, parent)
|
|
92
|
+
|
|
93
|
+
def _render_paragraph(self, paragraph) -> tuple[str, str] | None:
|
|
94
|
+
text = paragraph.text.strip()
|
|
95
|
+
if not text:
|
|
96
|
+
return None
|
|
97
|
+
|
|
98
|
+
style_name = paragraph.style.name.lower()
|
|
99
|
+
for needle, prefix in _HEADING_PREFIXES:
|
|
100
|
+
if needle in style_name:
|
|
101
|
+
return "heading", f"{prefix}{text}"
|
|
102
|
+
if "list" in style_name:
|
|
103
|
+
return "list_item", f"- {text}"
|
|
104
|
+
return "paragraph", text
|
|
105
|
+
|
|
106
|
+
def _table_rows(self, table) -> list[list[str]]:
|
|
107
|
+
return [[cell.text.strip().replace("\n", " ") for cell in row.cells] for row in table.rows]
|
|
108
|
+
|
|
109
|
+
def _rows_to_markdown(self, rows: list[list[str]]) -> str:
|
|
110
|
+
width = max(len(row) for row in rows)
|
|
111
|
+
padded = [row + [""] * (width - len(row)) for row in rows]
|
|
112
|
+
lines = [f"| {' | '.join(padded[0])} |", f"| {' | '.join(['---'] * width)} |"]
|
|
113
|
+
lines.extend(f"| {' | '.join(row)} |" for row in padded[1:])
|
|
114
|
+
return "\n".join(lines)
|