langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,134 @@
1
+ import threading
2
+ from collections.abc import Iterator
3
+ from pathlib import Path
4
+
5
+ from langparse.core.engine import BaseEngine, PageResult
6
+ from langparse.engines.pdf.ocr import (
7
+ DEFAULT_MIN_CHARS,
8
+ DEFAULT_RESOLUTION,
9
+ load_recogniser,
10
+ needs_ocr,
11
+ ocr_page_text,
12
+ )
13
+ from langparse.progress import ProgressCallback, ProgressReporter
14
+
15
+
16
+ class BasePDFEngine(BaseEngine):
17
+ """
18
+ Base class specifically for PDF engines.
19
+ """
20
+
21
+ def __init__(self, **kwargs):
22
+ pass
23
+
24
+
25
+ class SimplePDFEngine(BasePDFEngine):
26
+ """
27
+ A lightweight, dependency-free (except pdfplumber) engine.
28
+ Good for simple, native PDFs.
29
+
30
+ Scanned pages fall back to OCR. Without it a scanned document parses
31
+ "successfully" into a handful of watermark characters, which is harder for a
32
+ caller to notice than an outright failure.
33
+ """
34
+
35
+ def __init__(
36
+ self,
37
+ enable_ocr: bool = True,
38
+ ocr_min_chars: int = DEFAULT_MIN_CHARS,
39
+ ocr_resolution: int = DEFAULT_RESOLUTION,
40
+ recogniser=None,
41
+ **kwargs,
42
+ ):
43
+ self.enable_ocr = enable_ocr
44
+ self.ocr_min_chars = ocr_min_chars
45
+ self.ocr_resolution = ocr_resolution
46
+ self._recogniser = recogniser
47
+ # Batch runs share one engine across worker threads. The lock keeps
48
+ # model loading (tens of seconds) from happening once per thread, and
49
+ # serialises recognition because rapidocr states no thread-safety
50
+ # guarantee -- correctness over throughput on an already slow path.
51
+ self._ocr_lock = threading.Lock()
52
+
53
+ def process(
54
+ self, file_path: Path, *, progress_callback: ProgressCallback | None = None, **kwargs
55
+ ) -> Iterator[PageResult]:
56
+ reporter = ProgressReporter(str(file_path), progress_callback)
57
+ try:
58
+ import pdfplumber
59
+ except ImportError:
60
+ raise ImportError("Please install `pdfplumber` to use the 'simple' engine.") from None
61
+
62
+ with pdfplumber.open(file_path) as pdf:
63
+ for i, page in enumerate(pdf.pages):
64
+ text = page.extract_text() or ""
65
+ ocr_applied = False
66
+
67
+ if self.enable_ocr and needs_ocr(page, min_chars=self.ocr_min_chars):
68
+ recovered = self._run_ocr(page)
69
+ if recovered:
70
+ text = recovered
71
+ ocr_applied = True
72
+
73
+ tables, table_markdown = self._extract_tables(page)
74
+
75
+ markdown_content = text
76
+ if table_markdown:
77
+ markdown_content = "\n\n".join([text, "\n".join(table_markdown)]).strip()
78
+
79
+ reporter.emit(
80
+ "parsing", completed_units=i + 1, total_units=len(pdf.pages), unit="pages"
81
+ )
82
+ yield PageResult(
83
+ page_number=i + 1,
84
+ markdown_content=markdown_content,
85
+ plain_text=text,
86
+ elements=[],
87
+ tables=tables,
88
+ images=[],
89
+ metadata={
90
+ "engine_name": "simple",
91
+ "ocr_applied": ocr_applied,
92
+ "ocr_text_chars": len(text) if ocr_applied else 0,
93
+ },
94
+ )
95
+
96
+ def _run_ocr(self, page) -> str:
97
+ """Recognise a page, degrading to no text rather than failing the parse."""
98
+ try:
99
+ with self._ocr_lock:
100
+ recogniser = self._build_recogniser_if_needed()
101
+ return ocr_page_text(page, recogniser, resolution=self.ocr_resolution)
102
+ except ImportError:
103
+ # rapidocr absent: the page keeps whatever thin text layer it had.
104
+ return ""
105
+
106
+ def _resolve_recogniser(self):
107
+ with self._ocr_lock:
108
+ return self._build_recogniser_if_needed()
109
+
110
+ def _build_recogniser_if_needed(self):
111
+ """Caller must hold `_ocr_lock`."""
112
+ if self._recogniser is None:
113
+ self._recogniser = load_recogniser()
114
+ return self._recogniser
115
+
116
+ def _extract_tables(self, page):
117
+ tables = []
118
+ table_markdown = []
119
+ extract_tables = getattr(page, "extract_tables", None)
120
+ for table in (extract_tables() if callable(extract_tables) else []) or []:
121
+ cleaned_table = [
122
+ ["" if cell is None else str(cell).strip().replace("\n", " ") for cell in row]
123
+ for row in table
124
+ ]
125
+ if not cleaned_table:
126
+ continue
127
+
128
+ tables.append({"rows": cleaned_table})
129
+ headers = cleaned_table[0]
130
+ table_markdown.append(f"| {' | '.join(headers)} |")
131
+ table_markdown.append(f"| {' | '.join(['---'] * len(headers))} |")
132
+ for row in cleaned_table[1:]:
133
+ table_markdown.append(f"| {' | '.join(row)} |")
134
+ return tables, table_markdown
@@ -0,0 +1,27 @@
1
+ from collections.abc import Iterator
2
+ from pathlib import Path
3
+
4
+ from langparse.core.engine import PageResult
5
+ from langparse.engines.pdf.simple import BasePDFEngine
6
+ from langparse.logging import get_logger
7
+
8
+ logger = get_logger(__name__)
9
+
10
+
11
+ class VisionLLMEngine(BasePDFEngine):
12
+ """
13
+ Uses Vision LLMs (GPT-4o, Gemini 1.5 Pro) to parse pages.
14
+ Extremely slow/expensive but handles EVERYTHING (handwriting, complex charts).
15
+ """
16
+
17
+ def __init__(self, model_name: str = "gpt-4o", api_key: str = None):
18
+ self.model_name = model_name
19
+ self.api_key = api_key
20
+
21
+ def process(self, file_path: Path, **kwargs) -> Iterator[PageResult]:
22
+ # 1. Convert PDF to Images (using pdf2image)
23
+ # 2. Send each image to LLM with a prompt like "Transcribe this page to Markdown"
24
+
25
+ logger.debug("VisionLLM processing %s with %s", file_path, self.model_name)
26
+
27
+ raise NotImplementedError("Vision LLM integration is pending.")
langparse/errors.py ADDED
@@ -0,0 +1,70 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ from dataclasses import dataclass
5
+ from enum import Enum
6
+ from subprocess import TimeoutExpired
7
+ from urllib.error import URLError
8
+
9
+
10
+ class ErrorType(str, Enum):
11
+ DEPENDENCY_MISSING = "dependency_missing"
12
+ FILE_NOT_FOUND = "file_not_found"
13
+ UNSUPPORTED_FORMAT = "unsupported_format"
14
+ ENGINE_UNAVAILABLE = "engine_unavailable"
15
+ ENGINE_TIMEOUT = "engine_timeout"
16
+ PARSE_FAILED = "parse_failed"
17
+ QUALITY_CHECK_FAILED = "quality_check_failed"
18
+ OCR_UNAVAILABLE = "ocr_unavailable"
19
+ LAYOUT_QUALITY_WARNING = "layout_quality_warning"
20
+ TABLE_EXTRACTION_FAILED = "table_extraction_failed"
21
+ WORKBOOK_EVALUATION_FAILED = "workbook_evaluation_failed"
22
+
23
+
24
+ @dataclass
25
+ class ClassifiedError:
26
+ error_type: ErrorType
27
+ message: str
28
+
29
+
30
+ def classify_exception(exc: BaseException) -> ClassifiedError:
31
+ message = str(exc)
32
+ lowered = message.lower()
33
+
34
+ from langparse.workbooks.evaluation.schema import WorkbookEvaluationError
35
+
36
+ if isinstance(exc, WorkbookEvaluationError):
37
+ return ClassifiedError(ErrorType.WORKBOOK_EVALUATION_FAILED, message)
38
+ if isinstance(exc, FileNotFoundError):
39
+ return ClassifiedError(ErrorType.FILE_NOT_FOUND, message)
40
+ if isinstance(exc, ImportError):
41
+ return ClassifiedError(ErrorType.DEPENDENCY_MISSING, message)
42
+ if "unsupported file extension" in lowered:
43
+ return ClassifiedError(ErrorType.UNSUPPORTED_FORMAT, message)
44
+ if "cuda" in lowered and "not available" in lowered:
45
+ return ClassifiedError(ErrorType.ENGINE_UNAVAILABLE, message)
46
+ if "unable to start local mineru-api" in lowered:
47
+ return ClassifiedError(ErrorType.ENGINE_UNAVAILABLE, message)
48
+ if _is_timeout(exc) or re.search(r"\btimed out\b", lowered):
49
+ return ClassifiedError(ErrorType.ENGINE_TIMEOUT, message)
50
+ if "ocr" in lowered and "unavailable" in lowered:
51
+ return ClassifiedError(ErrorType.OCR_UNAVAILABLE, message)
52
+ if "table" in lowered and "failed" in lowered:
53
+ return ClassifiedError(ErrorType.TABLE_EXTRACTION_FAILED, message)
54
+
55
+ return ClassifiedError(ErrorType.PARSE_FAILED, message)
56
+
57
+
58
+ def _is_timeout(exc: BaseException) -> bool:
59
+ """Recognize standard timeout types through explicit backend wrappers."""
60
+ seen: set[int] = set()
61
+ current: BaseException | None = exc
62
+ while current is not None and id(current) not in seen:
63
+ seen.add(id(current))
64
+ if isinstance(current, (TimeoutError, TimeoutExpired)):
65
+ return True
66
+ if isinstance(current, URLError) and isinstance(current.reason, BaseException):
67
+ current = current.reason
68
+ else:
69
+ current = current.__cause__
70
+ return False
langparse/logging.py ADDED
@@ -0,0 +1,27 @@
1
+ """
2
+ Logging for the library.
3
+
4
+ A library should not choose a logging framework for the application embedding
5
+ it, nor write to stderr uninvited. The stdlib logger with a NullHandler stays
6
+ silent until the host configures logging, and costs no dependency -- which
7
+ leaves langparse installable with nothing but the extras a caller actually
8
+ needs for their formats.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import logging
14
+
15
+ ROOT_LOGGER_NAME = "langparse"
16
+
17
+ _root = logging.getLogger(ROOT_LOGGER_NAME)
18
+ _root.addHandler(logging.NullHandler())
19
+
20
+
21
+ def get_logger(name: str | None = None) -> logging.Logger:
22
+ """Return a logger under the `langparse` namespace."""
23
+ if not name:
24
+ return _root
25
+ if name.startswith(f"{ROOT_LOGGER_NAME}."):
26
+ return logging.getLogger(name)
27
+ return logging.getLogger(f"{ROOT_LOGGER_NAME}.{name}")
langparse/metrics.py ADDED
@@ -0,0 +1,129 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ from dataclasses import dataclass, field
5
+ from typing import Any
6
+
7
+ from langparse.types import Chunk, ParsedDocumentResult
8
+
9
+
10
+ def pages_per_second(page_count: int, elapsed_seconds: float) -> float:
11
+ if elapsed_seconds <= 0:
12
+ return 0.0
13
+ return round(page_count / elapsed_seconds, 4)
14
+
15
+
16
+ def count_markdown_tables(markdown: str) -> int:
17
+ separator_pattern = re.compile(r"^\s*\|?\s*:?-{3,}:?\s*(\|\s*:?-{3,}:?\s*)+\|?\s*$")
18
+ return sum(1 for line in markdown.splitlines() if separator_pattern.match(line))
19
+
20
+
21
+ @dataclass
22
+ class ParseMetrics:
23
+ elapsed_seconds: float = 0.0
24
+ page_count: int = 0
25
+ pages_per_second: float = 0.0
26
+ output_bytes: int = 0
27
+ markdown_chars: int = 0
28
+ table_count: int = 0
29
+ image_count: int = 0
30
+ chunk_count: int = 0
31
+ chunks_with_page_numbers_ratio: float = 0.0
32
+ page_marker_coverage: float = 0.0
33
+ ocr_applied: bool = False
34
+ ocr_text_chars: int = 0
35
+ multi_column_detected: bool = False
36
+ reading_order_warnings: int = 0
37
+ header_footer_removed_count: int = 0
38
+ caption_count: int = 0
39
+ images_with_caption_ratio: float = 0.0
40
+
41
+
42
+ @dataclass
43
+ class BatchItemResult:
44
+ source: str
45
+ status: str
46
+ output_path: str | None = None
47
+ metrics: ParseMetrics | None = None
48
+ error_type: str | None = None
49
+ error_message: str | None = None
50
+ engine: str | None = None
51
+ started_at: str | None = None
52
+ finished_at: str | None = None
53
+
54
+
55
+ @dataclass
56
+ class BatchRunResult:
57
+ items: list[BatchItemResult] = field(default_factory=list)
58
+ summary: dict[str, Any] = field(default_factory=dict)
59
+ #: Rendered documents, populated only when the run had no output directory
60
+ #: and therefore rendered to memory instead of to disk. Kept off
61
+ #: BatchItemResult so it never bloats the JSONL report.
62
+ rendered_outputs: list[str] = field(default_factory=list)
63
+
64
+ @property
65
+ def total_files(self) -> int:
66
+ return len(self.items)
67
+
68
+ @property
69
+ def success_count(self) -> int:
70
+ return sum(1 for item in self.items if item.status == "success")
71
+
72
+ @property
73
+ def failed_count(self) -> int:
74
+ return sum(1 for item in self.items if item.status == "failed")
75
+
76
+ @property
77
+ def skipped_count(self) -> int:
78
+ return sum(1 for item in self.items if item.status == "skipped")
79
+
80
+
81
+ def page_marker_coverage(parsed: ParsedDocumentResult) -> float:
82
+ """
83
+ Fraction of pages carrying a usable page number.
84
+
85
+ This is what makes a chunk citable back to a page, so the
86
+ ``require_page_markers`` quality check reads it directly.
87
+ """
88
+ if not getattr(parsed, "paginated", True) or not parsed.pages:
89
+ return 0.0
90
+ numbered = sum(1 for page in parsed.pages if page.page_number and page.page_number >= 1)
91
+ return round(numbered / len(parsed.pages), 4)
92
+
93
+
94
+ def collect_parse_metrics(
95
+ parsed: ParsedDocumentResult,
96
+ elapsed_seconds: float,
97
+ chunks: list[Chunk] | None = None,
98
+ ) -> ParseMetrics:
99
+ markdown = parsed.markdown_content or ""
100
+ page_count = len(parsed.pages)
101
+ chunk_count = len(chunks) if chunks is not None else 0
102
+ chunks_with_pages = (
103
+ sum(1 for chunk in chunks if chunk.metadata.get("page_numbers")) if chunks else 0
104
+ )
105
+ image_count = sum(len(page.images) for page in parsed.pages)
106
+ table_count = sum(len(page.tables) for page in parsed.pages) or count_markdown_tables(markdown)
107
+ caption_count = sum(1 for page in parsed.pages for image in page.images if image.get("caption"))
108
+
109
+ return ParseMetrics(
110
+ elapsed_seconds=round(elapsed_seconds, 4),
111
+ page_count=page_count,
112
+ pages_per_second=pages_per_second(page_count, elapsed_seconds),
113
+ output_bytes=len(markdown.encode("utf-8")),
114
+ markdown_chars=len(markdown),
115
+ table_count=table_count,
116
+ image_count=image_count,
117
+ chunk_count=chunk_count,
118
+ chunks_with_page_numbers_ratio=round(chunks_with_pages / chunk_count, 4)
119
+ if chunk_count
120
+ else 0.0,
121
+ page_marker_coverage=page_marker_coverage(parsed),
122
+ ocr_applied=bool(parsed.metadata.get("ocr_applied", False)),
123
+ ocr_text_chars=int(parsed.metadata.get("ocr_text_chars", 0) or 0),
124
+ multi_column_detected=bool(parsed.metadata.get("multi_column_detected", False)),
125
+ reading_order_warnings=int(parsed.metadata.get("reading_order_warnings", 0) or 0),
126
+ header_footer_removed_count=int(parsed.metadata.get("header_footer_removed_count", 0) or 0),
127
+ caption_count=caption_count,
128
+ images_with_caption_ratio=round(caption_count / image_count, 4) if image_count else 0.0,
129
+ )
File without changes
@@ -0,0 +1,114 @@
1
+ from pathlib import Path
2
+
3
+ from langparse.core.parser import BaseParser
4
+ from langparse.types import ParsedDocumentResult, ParsedElement, ParsedPageResult
5
+
6
+ _HEADING_PREFIXES = (
7
+ ("heading 1", "# "),
8
+ ("title", "# "),
9
+ ("heading 2", "## "),
10
+ ("heading 3", "### "),
11
+ )
12
+
13
+
14
+ class DocxParser(BaseParser):
15
+ """
16
+ Parses .docx files to Markdown.
17
+ Note: DOCX is a flow format, so 'page numbers' are not strictly defined.
18
+ We treat the entire document as Page 1 for now, unless we convert to PDF first.
19
+ """
20
+
21
+ def parse_result(self, file_path: str | Path, **kwargs) -> ParsedDocumentResult:
22
+ path = self._resolve_existing_path(file_path)
23
+
24
+ try:
25
+ import docx
26
+ except ImportError:
27
+ raise ImportError(
28
+ "python-docx is required. Install with `pip install python-docx`."
29
+ ) from None
30
+
31
+ doc = docx.Document(path)
32
+ markdown_lines: list[str] = []
33
+ elements: list[ParsedElement] = []
34
+ tables: list[dict] = []
35
+
36
+ for block in self._iter_block_items(docx, doc):
37
+ if isinstance(block, docx.text.paragraph.Paragraph):
38
+ rendered = self._render_paragraph(block)
39
+ if rendered is None:
40
+ continue
41
+ kind, markdown = rendered
42
+ markdown_lines.extend([markdown, ""])
43
+ elements.append(ParsedElement(kind=kind, text=block.text.strip()))
44
+ continue
45
+
46
+ rows = self._table_rows(block)
47
+ if not rows:
48
+ continue
49
+ table_markdown = self._rows_to_markdown(rows)
50
+ tables.append({"rows": rows})
51
+ markdown_lines.extend([table_markdown, ""])
52
+ elements.append(ParsedElement(kind="table", text=table_markdown))
53
+
54
+ markdown_content = "\n".join(markdown_lines)
55
+ return ParsedDocumentResult(
56
+ source=str(path),
57
+ filename=path.name,
58
+ engine="docx",
59
+ pages=[
60
+ ParsedPageResult(
61
+ page_number=1,
62
+ markdown_content=markdown_content,
63
+ plain_text="\n".join(
64
+ element.text for element in elements if element.kind != "table"
65
+ ),
66
+ elements=elements,
67
+ tables=tables,
68
+ )
69
+ ],
70
+ markdown_content=markdown_content,
71
+ metadata={"extension": ".docx"},
72
+ )
73
+
74
+ def _iter_block_items(self, docx, parent):
75
+ """
76
+ Yield paragraphs and tables in document order.
77
+
78
+ python-docx exposes paragraphs and tables as separate collections, which
79
+ loses their interleaving, so walk the body XML directly.
80
+ """
81
+ from docx.oxml.table import CT_Tbl
82
+ from docx.oxml.text.paragraph import CT_P
83
+
84
+ parent_elm = (
85
+ parent.element.body if isinstance(parent, docx.document.Document) else parent._element
86
+ )
87
+ for child in parent_elm.iterchildren():
88
+ if isinstance(child, CT_P):
89
+ yield docx.text.paragraph.Paragraph(child, parent)
90
+ elif isinstance(child, CT_Tbl):
91
+ yield docx.table.Table(child, parent)
92
+
93
+ def _render_paragraph(self, paragraph) -> tuple[str, str] | None:
94
+ text = paragraph.text.strip()
95
+ if not text:
96
+ return None
97
+
98
+ style_name = paragraph.style.name.lower()
99
+ for needle, prefix in _HEADING_PREFIXES:
100
+ if needle in style_name:
101
+ return "heading", f"{prefix}{text}"
102
+ if "list" in style_name:
103
+ return "list_item", f"- {text}"
104
+ return "paragraph", text
105
+
106
+ def _table_rows(self, table) -> list[list[str]]:
107
+ return [[cell.text.strip().replace("\n", " ") for cell in row.cells] for row in table.rows]
108
+
109
+ def _rows_to_markdown(self, rows: list[list[str]]) -> str:
110
+ width = max(len(row) for row in rows)
111
+ padded = [row + [""] * (width - len(row)) for row in rows]
112
+ lines = [f"| {' | '.join(padded[0])} |", f"| {' | '.join(['---'] * width)} |"]
113
+ lines.extend(f"| {' | '.join(row)} |" for row in padded[1:])
114
+ return "\n".join(lines)