universal-doc-parser 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. universal_doc_parser-1.0.0.dist-info/METADATA +692 -0
  2. universal_doc_parser-1.0.0.dist-info/RECORD +48 -0
  3. universal_doc_parser-1.0.0.dist-info/WHEEL +4 -0
  4. universal_doc_parser-1.0.0.dist-info/licenses/LICENSE +21 -0
  5. universal_parser/__init__.py +28 -0
  6. universal_parser/adaptive/__init__.py +17 -0
  7. universal_parser/adaptive/config_cache.py +127 -0
  8. universal_parser/adaptive/fingerprint.py +133 -0
  9. universal_parser/adaptive/tuner.py +97 -0
  10. universal_parser/core/__init__.py +0 -0
  11. universal_parser/core/engine.py +55 -0
  12. universal_parser/core/router.py +38 -0
  13. universal_parser/core/schema.py +58 -0
  14. universal_parser/core/sniffer.py +125 -0
  15. universal_parser/enrichment/__init__.py +0 -0
  16. universal_parser/enrichment/vlm_enricher.py +0 -0
  17. universal_parser/exports/__init__.py +1 -0
  18. universal_parser/exports/to_chunks.py +108 -0
  19. universal_parser/exports/to_graph.py +114 -0
  20. universal_parser/exports/to_markdown.py +31 -0
  21. universal_parser/extractors/__init__.py +0 -0
  22. universal_parser/extractors/base.py +39 -0
  23. universal_parser/extractors/images/__init__.py +0 -0
  24. universal_parser/extractors/images/scan_extractor.py +79 -0
  25. universal_parser/extractors/mail/__init__.py +0 -0
  26. universal_parser/extractors/mail/mail_extractor.py +206 -0
  27. universal_parser/extractors/office/__init__.py +0 -0
  28. universal_parser/extractors/office/docx_extractor.py +131 -0
  29. universal_parser/extractors/office/legacy_extractor.py +112 -0
  30. universal_parser/extractors/office/pptx_extractor.py +109 -0
  31. universal_parser/extractors/office/xlsx_extractor.py +93 -0
  32. universal_parser/extractors/pdf/__init__.py +0 -0
  33. universal_parser/extractors/pdf/native.py +331 -0
  34. universal_parser/extractors/pdf/tables.py +155 -0
  35. universal_parser/extractors/pdf/visual_onnx.py +0 -0
  36. universal_parser/extractors/structured/__init__.py +0 -0
  37. universal_parser/extractors/structured/csv_extractor.py +86 -0
  38. universal_parser/extractors/structured/json_xml_extractor.py +104 -0
  39. universal_parser/extractors/structured/parquet_extractor.py +59 -0
  40. universal_parser/extractors/web/__init__.py +1 -0
  41. universal_parser/extractors/web/epub_extractor.py +81 -0
  42. universal_parser/extractors/web/html_extractor.py +111 -0
  43. universal_parser/mcp/__init__.py +1 -0
  44. universal_parser/mcp/server.py +61 -0
  45. universal_parser/observability/__init__.py +12 -0
  46. universal_parser/observability/dashboard.py +231 -0
  47. universal_parser/observability/logger.py +0 -0
  48. universal_parser/observability/metrics.py +99 -0
@@ -0,0 +1,125 @@
1
+ from __future__ import annotations
2
+
3
+ from enum import Enum, auto
4
+ from pathlib import Path
5
+
6
+ try:
7
+ import magic
8
+ except (ImportError, Exception):
9
+ magic = None
10
+
11
+
12
+
13
+ class FileType(Enum):
14
+ """All document types this parser understands."""
15
+
16
+ PDF = auto()
17
+ DOCX = auto()
18
+ XLSX = auto()
19
+ PPTX = auto()
20
+ DOC = auto() # legacy binary — Phase 6
21
+ XLS = auto() # legacy binary — Phase 6
22
+ PPT = auto() # legacy binary — Phase 6
23
+ HTML = auto()
24
+ EPUB = auto()
25
+ CSV = auto()
26
+ TSV = auto()
27
+ PARQUET = auto()
28
+ JSON = auto()
29
+ XML = auto()
30
+ EML = auto()
31
+ MSG = auto()
32
+ MBOX = auto()
33
+ IMAGE = auto() # tiff, bmp, webp, jpg, png
34
+ UNKNOWN = auto() # never crash — return this for anything unrecognized
35
+
36
+
37
+ _EXT_MAP: dict[str, FileType] = {
38
+ ".pdf": FileType.PDF,
39
+ ".docx": FileType.DOCX,
40
+ ".xlsx": FileType.XLSX,
41
+ ".pptx": FileType.PPTX,
42
+ ".doc": FileType.DOC,
43
+ ".xls": FileType.XLS,
44
+ ".ppt": FileType.PPT,
45
+ ".html": FileType.HTML,
46
+ ".htm": FileType.HTML,
47
+ ".xhtml": FileType.HTML,
48
+ ".epub": FileType.EPUB,
49
+ ".csv": FileType.CSV,
50
+ ".tsv": FileType.TSV,
51
+ ".parquet": FileType.PARQUET,
52
+ ".json": FileType.JSON,
53
+ ".xml": FileType.XML,
54
+ ".eml": FileType.EML,
55
+ ".msg": FileType.MSG,
56
+ ".mbox": FileType.MBOX,
57
+ ".tiff": FileType.IMAGE,
58
+ ".tif": FileType.IMAGE,
59
+ ".bmp": FileType.IMAGE,
60
+ ".webp": FileType.IMAGE,
61
+ ".jpg": FileType.IMAGE,
62
+ ".jpeg": FileType.IMAGE,
63
+ ".png": FileType.IMAGE,
64
+ }
65
+
66
+ _MIME_MAP: dict[str, FileType] = {
67
+ "application/pdf": FileType.PDF,
68
+ "application/vnd.openxmlformats-officedocument.wordprocessingml.document": FileType.DOCX,
69
+ "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet": FileType.XLSX,
70
+ "application/vnd.openxmlformats-officedocument.presentationml.presentation": FileType.PPTX,
71
+ "application/msword": FileType.DOC,
72
+ "application/vnd.ms-excel": FileType.XLS,
73
+ "application/vnd.ms-powerpoint": FileType.PPT,
74
+ "text/html": FileType.HTML,
75
+ "application/epub+zip": FileType.EPUB,
76
+ "text/csv": FileType.CSV,
77
+ "text/plain": FileType.CSV, # resolved by extension
78
+ "application/json": FileType.JSON,
79
+ "text/xml": FileType.XML,
80
+ "application/xml": FileType.XML,
81
+ "message/rfc822": FileType.EML,
82
+ "image/tiff": FileType.IMAGE,
83
+ "image/bmp": FileType.IMAGE,
84
+ "image/webp": FileType.IMAGE,
85
+ "image/jpeg": FileType.IMAGE,
86
+ "image/png": FileType.IMAGE,
87
+ }
88
+
89
+
90
+ def sniff(path: str | Path) -> FileType:
91
+ """
92
+ Detect the file type of the given path.
93
+ Strategy:
94
+ 1. Read magic bytes via python-magic -> look up MIME in _MIME_MAP
95
+ 2. If ambiguous (e.g. text/plain could be CSV or TSV), fall back to extension
96
+ 3. If still unknown, return FileType.UNKNOWN — never raise
97
+ Args:
98
+ path: path to any file
99
+ Returns:
100
+ FileType enum value
101
+ """
102
+ path = Path(path)
103
+ ext = path.suffix.lower()
104
+
105
+ if magic is not None:
106
+ try:
107
+ mime = magic.from_file(str(path), mime=True)
108
+ file_type = _MIME_MAP.get(mime)
109
+
110
+ # MIME was recognized but ambiguous — let extension break the tie
111
+ if file_type in (FileType.CSV, FileType.HTML, None):
112
+ ext_type = _EXT_MAP.get(ext)
113
+ if ext_type is not None:
114
+ return ext_type
115
+
116
+ if file_type is not None:
117
+ return file_type
118
+
119
+ except Exception: # noqa: BLE001, S110
120
+ # magic can fail on locked files, permission errors, etc.
121
+ # fall through to extension lookup
122
+ pass
123
+
124
+ # Last resort: extension only
125
+ return _EXT_MAP.get(ext, FileType.UNKNOWN)
File without changes
File without changes
@@ -0,0 +1 @@
1
+ # Export package
@@ -0,0 +1,108 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass, field
4
+
5
+ from universal_parser.core.schema import Document
6
+
7
+
8
+ @dataclass
9
+ class Chunk:
10
+ """A token-budgeted chunk ready for vector database embeddings."""
11
+
12
+ chunk_id: str
13
+ text: str
14
+ headings: list[str] = field(default_factory=list)
15
+ element_types: list[str] = field(default_factory=list)
16
+ page_numbers: list[int] = field(default_factory=list)
17
+ estimated_tokens: int = 0
18
+
19
+
20
+ def to_chunks(
21
+ doc: Document,
22
+ max_tokens: int = 512,
23
+ overlap_tokens: int = 50,
24
+ ) -> list[Chunk]:
25
+ """
26
+ Hierarchical token-aware chunker for RAG pipelines.
27
+ Rules:
28
+ - Tracks active H1, H2, H3 hierarchy across elements
29
+ - Prefixes chunk with current heading path context
30
+ - Bundles elements until max_tokens budget is reached
31
+ - Keeps tables intact within a single chunk where possible
32
+ """
33
+ chunks: list[Chunk] = []
34
+ current_headings: dict[int, str] = {} # level -> heading text
35
+ current_elements: list[str] = []
36
+ current_types: list[int] = []
37
+ current_pages: list[int] = []
38
+ current_tokens = 0
39
+ chunk_index = 1
40
+
41
+ for el in doc.content_tree:
42
+ # 1. Update heading stack
43
+ if el.type == "heading":
44
+ level = el.level or 1
45
+ current_headings = {lvl: txt for lvl, txt in current_headings.items() if lvl < level}
46
+ current_headings[level] = el.text or ""
47
+
48
+ # 2. Get text representation
49
+ el_text = el.markdown_repr or el.text or ""
50
+ if not el_text.strip():
51
+ continue
52
+
53
+ # Fast token estimation (~4 chars per token)
54
+ el_tokens = max(1, len(el_text) // 4)
55
+
56
+ # 3. Check if adding exceeds budget
57
+ if current_tokens + el_tokens > max_tokens and current_elements:
58
+ heading_hierarchy = [txt for _, txt in sorted(current_headings.items())]
59
+ heading_prefix = " > ".join(heading_hierarchy)
60
+ header_str = f"Context: {heading_prefix}\n\n" if heading_prefix else ""
61
+ chunk_text = header_str + "\n\n".join(current_elements)
62
+ total_est = len(chunk_text) // 4
63
+
64
+ chunks.append(
65
+ Chunk(
66
+ chunk_id=f"{doc.doc_id}-chunk-{chunk_index}",
67
+ text=chunk_text,
68
+ headings=heading_hierarchy,
69
+ element_types=list(set(current_types)),
70
+ page_numbers=sorted(set(current_pages)),
71
+ estimated_tokens=total_est,
72
+ )
73
+ )
74
+ chunk_index += 1
75
+
76
+ current_elements = []
77
+ current_types = []
78
+ current_pages = []
79
+ current_tokens = 0
80
+
81
+ # 4. Append element to current buffer
82
+ current_elements.append(el_text)
83
+ current_types.append(el.type)
84
+ if el.page is not None:
85
+ current_pages.append(el.page)
86
+ current_tokens += el_tokens
87
+
88
+ # Flush remaining buffer
89
+ if current_elements:
90
+ heading_hierarchy = [txt for _, txt in sorted(current_headings.items())]
91
+ heading_prefix = " > ".join(heading_hierarchy)
92
+ header_str = f"Context: {heading_prefix}\n\n" if heading_prefix else ""
93
+
94
+ chunk_text = header_str + "\n\n".join(current_elements)
95
+ total_est = len(chunk_text) // 4
96
+
97
+ chunks.append(
98
+ Chunk(
99
+ chunk_id=f"{doc.doc_id}-chunk-{chunk_index}",
100
+ text=chunk_text,
101
+ headings=heading_hierarchy,
102
+ element_types=list(set(current_types)),
103
+ page_numbers=sorted(set(current_pages)),
104
+ estimated_tokens=total_est,
105
+ )
106
+ )
107
+
108
+ return chunks
@@ -0,0 +1,114 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass, field
4
+
5
+ from universal_parser.core.schema import Document
6
+
7
+
8
+ @dataclass
9
+ class GraphNode:
10
+ id: str
11
+ type: str # "document", "section", "table", "content"
12
+ properties: dict[str, str | int | float] = field(default_factory=dict)
13
+
14
+
15
+ @dataclass
16
+ class GraphEdge:
17
+ source_id: str
18
+ target_id: str
19
+ relationship: str # "CONTAINS_SECTION", "CONTAINS", "FOLLOWS"
20
+
21
+
22
+ @dataclass
23
+ class KnowledgeGraph:
24
+ nodes: list[GraphNode] = field(default_factory=list)
25
+ edges: list[GraphEdge] = field(default_factory=list)
26
+
27
+
28
+ def to_graph(doc: Document) -> KnowledgeGraph:
29
+ """
30
+ Export Document into a Knowledge Graph structure for GraphRAG & Neo4j.
31
+
32
+ Creates:
33
+ - Document Root Node
34
+ - Heading / Section Nodes with hierarchical edges
35
+ - Paragraph & Table Content Nodes linked to their parent sections
36
+ """
37
+ nodes: list[GraphNode] = []
38
+ edges: list[GraphEdge] = []
39
+
40
+ # 1. Document Root Node
41
+ doc_node_id = f"doc:{doc.doc_id}"
42
+ nodes.append(
43
+ GraphNode(
44
+ id=doc_node_id,
45
+ type="document",
46
+ properties={
47
+ "file_name": doc.metadata.file_name,
48
+ "file_type": doc.metadata.file_type,
49
+ },
50
+ )
51
+ )
52
+
53
+ heading_stack: list[tuple[int, str]] = [] # (level, node_id)
54
+ prev_node_id: str | None = None
55
+
56
+ for idx, el in enumerate(doc.content_tree, start=1):
57
+ node_id = f"el:{doc.doc_id}:{idx}"
58
+
59
+ if el.type == "heading":
60
+ level = el.level or 1
61
+ while heading_stack and heading_stack[-1][0] >= level:
62
+ heading_stack.pop()
63
+
64
+ parent_id = heading_stack[-1][1] if heading_stack else doc_node_id
65
+
66
+ nodes.append(
67
+ GraphNode(
68
+ id=node_id,
69
+ type="section",
70
+ properties={"title": el.text or "", "level": level},
71
+ )
72
+ )
73
+ edges.append(
74
+ GraphEdge(
75
+ source_id=parent_id,
76
+ target_id=node_id,
77
+ relationship="CONTAINS_SECTION",
78
+ )
79
+ )
80
+ heading_stack.append((level, node_id))
81
+
82
+ else:
83
+ parent_id = heading_stack[-1][1] if heading_stack else doc_node_id
84
+ node_type = "table" if el.type == "table" else "content"
85
+
86
+ nodes.append(
87
+ GraphNode(
88
+ id=node_id,
89
+ type=node_type,
90
+ properties={
91
+ "text": el.text or el.markdown_repr or "",
92
+ "element_type": el.type,
93
+ },
94
+ )
95
+ )
96
+ edges.append(
97
+ GraphEdge(
98
+ source_id=parent_id,
99
+ target_id=node_id,
100
+ relationship="CONTAINS",
101
+ )
102
+ )
103
+
104
+ if prev_node_id:
105
+ edges.append(
106
+ GraphEdge(
107
+ source_id=prev_node_id,
108
+ target_id=node_id,
109
+ relationship="FOLLOWS",
110
+ )
111
+ )
112
+ prev_node_id = node_id
113
+
114
+ return KnowledgeGraph(nodes=nodes, edges=edges)
@@ -0,0 +1,31 @@
1
+ from __future__ import annotations
2
+
3
+ from universal_parser.core.schema import Document
4
+
5
+
6
+ def to_markdown(doc: Document) -> str:
7
+ """
8
+ Convert a parsed Document into a clean, unified Markdown string.
9
+ Rules:
10
+ - Headings render as # H1, ## H2, ### H3
11
+ - Paragraphs render as clean prose
12
+ - List items render with "- "
13
+ - Tables render with structured Markdown grids
14
+ - Code blocks render within ``` fences
15
+ """
16
+ blocks: list[str] = []
17
+
18
+ for el in doc.content_tree:
19
+ if el.markdown_repr:
20
+ blocks.append(el.markdown_repr)
21
+ elif el.type == "heading":
22
+ level = el.level or 1
23
+ blocks.append(f"{'#' * level} {el.text or ''}")
24
+ elif el.type == "list_item":
25
+ blocks.append(f"- {el.text or ''}")
26
+ elif el.type == "code_block":
27
+ blocks.append(f"```\n{el.text or ''}\n```")
28
+ elif el.text:
29
+ blocks.append(el.text)
30
+
31
+ return "\n\n".join(blocks)
File without changes
@@ -0,0 +1,39 @@
1
+ from __future__ import annotations
2
+
3
+ from abc import ABC, abstractmethod
4
+ from collections.abc import Iterator
5
+ from pathlib import Path
6
+ from typing import ClassVar
7
+
8
+ from universal_parser.core.schema import Element
9
+ from universal_parser.core.sniffer import FileType
10
+
11
+
12
+ class BaseExtractor(ABC):
13
+ """
14
+ Abstract base class for all format extractors.
15
+ Every extractor in universal_parser.extractors must:
16
+ 1. Inherit from BaseExtractor
17
+ 2. Declare which FileTypes it handles via supported_types
18
+ 3. Implement stream() — yield Elements one at a time, never return a list
19
+ Adding a new format = one new file inheriting this + one entry in router.py.
20
+ Nothing else in core/ needs to change.
21
+ """
22
+
23
+ supported_types: ClassVar[list[FileType]] = []
24
+
25
+ @abstractmethod
26
+ def stream(self, path: str | Path) -> Iterator[Element]:
27
+ """
28
+ Parse the document at the given path and yield Element objects.
29
+ Rules:
30
+ - Must yield, never return a full list (streaming = memory safe)
31
+ - Must never raise on a recoverable error — yield a low-confidence
32
+ element or skip, log the issue, keep going
33
+ - One Element at a time, in reading order
34
+ Args:
35
+ path: absolute or relative path to the source file
36
+ Yields:
37
+ Element objects in document reading order
38
+ """
39
+ ...
File without changes
@@ -0,0 +1,79 @@
1
+ from __future__ import annotations
2
+
3
+ from collections.abc import Iterator
4
+ from pathlib import Path
5
+ from typing import ClassVar
6
+
7
+ import cv2
8
+ import numpy as np
9
+ from rapidocr_onnxruntime import RapidOCR
10
+
11
+ from universal_parser.core.router import register
12
+ from universal_parser.core.schema import BBox, Element
13
+ from universal_parser.core.sniffer import FileType
14
+ from universal_parser.extractors.base import BaseExtractor
15
+
16
+
17
+ @register
18
+ class ImageScanExtractor(BaseExtractor):
19
+ """
20
+ Extractor for image and scanned files (.png, .jpg, .tiff, .bmp, .webp).
21
+ Handles:
22
+ - Image preprocessing (deskewing / contrast enhancement via OpenCV)
23
+ - CPU-based text and bounding box detection via RapidOCR (ONNX)
24
+ - Confidence scoring per text element
25
+ """
26
+
27
+ supported_types: ClassVar[list[FileType]] = [FileType.IMAGE]
28
+
29
+ def __init__(self) -> None:
30
+ super().__init__()
31
+ # Initialize RapidOCR engine (cached in memory)
32
+ self._ocr = RapidOCR()
33
+
34
+ def stream(self, path: str | Path) -> Iterator[Element]:
35
+ path_str = str(path)
36
+
37
+ try:
38
+ # Step 1: Read image with OpenCV
39
+ img = cv2.imread(path_str)
40
+ if img is None:
41
+ return
42
+
43
+ # Step 2: Preprocess / deskew
44
+ preprocessing_img = self._preprocess_image(img)
45
+
46
+ # Step 3: Run RapidOCR
47
+ ocr_results, _ = self._ocr(preprocessing_img)
48
+ if not ocr_results:
49
+ return
50
+ # Step 4: Stream extracted elements
51
+ for item in ocr_results:
52
+ dt_boxes, text, score = item
53
+ clean_text = text.strip()
54
+ if not clean_text:
55
+ continue
56
+
57
+ # Compute bounding box (x0, y0, x1, y1)
58
+ pts = np.array(dt_boxes, dtype=np.float32)
59
+ x0 = float(np.min(pts[:, 0]))
60
+ y0 = float(np.min(pts[:, 1]))
61
+ x1 = float(np.max(pts[:, 0]))
62
+ y1 = float(np.max(pts[:, 1]))
63
+
64
+ yield Element(
65
+ type="paragraph",
66
+ text=clean_text,
67
+ page=1,
68
+ bbox=BBox(x0=x0, y0=y0, x1=x1, y1=y1),
69
+ markdown_repr=clean_text,
70
+ confidence=round(float(score), 3),
71
+ )
72
+
73
+ except Exception: # noqa: BLE001
74
+ return
75
+
76
+ def _preprocess_image(self, img: np.ndarray) -> np.ndarray:
77
+ """Basic contrast & grayscale cleanup for sharper OCR."""
78
+ # RapidOCR handles RGB internally; return original if valid
79
+ return img
File without changes