universal-doc-parser 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- universal_doc_parser-1.0.0.dist-info/METADATA +692 -0
- universal_doc_parser-1.0.0.dist-info/RECORD +48 -0
- universal_doc_parser-1.0.0.dist-info/WHEEL +4 -0
- universal_doc_parser-1.0.0.dist-info/licenses/LICENSE +21 -0
- universal_parser/__init__.py +28 -0
- universal_parser/adaptive/__init__.py +17 -0
- universal_parser/adaptive/config_cache.py +127 -0
- universal_parser/adaptive/fingerprint.py +133 -0
- universal_parser/adaptive/tuner.py +97 -0
- universal_parser/core/__init__.py +0 -0
- universal_parser/core/engine.py +55 -0
- universal_parser/core/router.py +38 -0
- universal_parser/core/schema.py +58 -0
- universal_parser/core/sniffer.py +125 -0
- universal_parser/enrichment/__init__.py +0 -0
- universal_parser/enrichment/vlm_enricher.py +0 -0
- universal_parser/exports/__init__.py +1 -0
- universal_parser/exports/to_chunks.py +108 -0
- universal_parser/exports/to_graph.py +114 -0
- universal_parser/exports/to_markdown.py +31 -0
- universal_parser/extractors/__init__.py +0 -0
- universal_parser/extractors/base.py +39 -0
- universal_parser/extractors/images/__init__.py +0 -0
- universal_parser/extractors/images/scan_extractor.py +79 -0
- universal_parser/extractors/mail/__init__.py +0 -0
- universal_parser/extractors/mail/mail_extractor.py +206 -0
- universal_parser/extractors/office/__init__.py +0 -0
- universal_parser/extractors/office/docx_extractor.py +131 -0
- universal_parser/extractors/office/legacy_extractor.py +112 -0
- universal_parser/extractors/office/pptx_extractor.py +109 -0
- universal_parser/extractors/office/xlsx_extractor.py +93 -0
- universal_parser/extractors/pdf/__init__.py +0 -0
- universal_parser/extractors/pdf/native.py +331 -0
- universal_parser/extractors/pdf/tables.py +155 -0
- universal_parser/extractors/pdf/visual_onnx.py +0 -0
- universal_parser/extractors/structured/__init__.py +0 -0
- universal_parser/extractors/structured/csv_extractor.py +86 -0
- universal_parser/extractors/structured/json_xml_extractor.py +104 -0
- universal_parser/extractors/structured/parquet_extractor.py +59 -0
- universal_parser/extractors/web/__init__.py +1 -0
- universal_parser/extractors/web/epub_extractor.py +81 -0
- universal_parser/extractors/web/html_extractor.py +111 -0
- universal_parser/mcp/__init__.py +1 -0
- universal_parser/mcp/server.py +61 -0
- universal_parser/observability/__init__.py +12 -0
- universal_parser/observability/dashboard.py +231 -0
- universal_parser/observability/logger.py +0 -0
- universal_parser/observability/metrics.py +99 -0
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from enum import Enum, auto
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
try:
|
|
7
|
+
import magic
|
|
8
|
+
except (ImportError, Exception):
|
|
9
|
+
magic = None
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class FileType(Enum):
|
|
14
|
+
"""All document types this parser understands."""
|
|
15
|
+
|
|
16
|
+
PDF = auto()
|
|
17
|
+
DOCX = auto()
|
|
18
|
+
XLSX = auto()
|
|
19
|
+
PPTX = auto()
|
|
20
|
+
DOC = auto() # legacy binary — Phase 6
|
|
21
|
+
XLS = auto() # legacy binary — Phase 6
|
|
22
|
+
PPT = auto() # legacy binary — Phase 6
|
|
23
|
+
HTML = auto()
|
|
24
|
+
EPUB = auto()
|
|
25
|
+
CSV = auto()
|
|
26
|
+
TSV = auto()
|
|
27
|
+
PARQUET = auto()
|
|
28
|
+
JSON = auto()
|
|
29
|
+
XML = auto()
|
|
30
|
+
EML = auto()
|
|
31
|
+
MSG = auto()
|
|
32
|
+
MBOX = auto()
|
|
33
|
+
IMAGE = auto() # tiff, bmp, webp, jpg, png
|
|
34
|
+
UNKNOWN = auto() # never crash — return this for anything unrecognized
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
_EXT_MAP: dict[str, FileType] = {
|
|
38
|
+
".pdf": FileType.PDF,
|
|
39
|
+
".docx": FileType.DOCX,
|
|
40
|
+
".xlsx": FileType.XLSX,
|
|
41
|
+
".pptx": FileType.PPTX,
|
|
42
|
+
".doc": FileType.DOC,
|
|
43
|
+
".xls": FileType.XLS,
|
|
44
|
+
".ppt": FileType.PPT,
|
|
45
|
+
".html": FileType.HTML,
|
|
46
|
+
".htm": FileType.HTML,
|
|
47
|
+
".xhtml": FileType.HTML,
|
|
48
|
+
".epub": FileType.EPUB,
|
|
49
|
+
".csv": FileType.CSV,
|
|
50
|
+
".tsv": FileType.TSV,
|
|
51
|
+
".parquet": FileType.PARQUET,
|
|
52
|
+
".json": FileType.JSON,
|
|
53
|
+
".xml": FileType.XML,
|
|
54
|
+
".eml": FileType.EML,
|
|
55
|
+
".msg": FileType.MSG,
|
|
56
|
+
".mbox": FileType.MBOX,
|
|
57
|
+
".tiff": FileType.IMAGE,
|
|
58
|
+
".tif": FileType.IMAGE,
|
|
59
|
+
".bmp": FileType.IMAGE,
|
|
60
|
+
".webp": FileType.IMAGE,
|
|
61
|
+
".jpg": FileType.IMAGE,
|
|
62
|
+
".jpeg": FileType.IMAGE,
|
|
63
|
+
".png": FileType.IMAGE,
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
_MIME_MAP: dict[str, FileType] = {
|
|
67
|
+
"application/pdf": FileType.PDF,
|
|
68
|
+
"application/vnd.openxmlformats-officedocument.wordprocessingml.document": FileType.DOCX,
|
|
69
|
+
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet": FileType.XLSX,
|
|
70
|
+
"application/vnd.openxmlformats-officedocument.presentationml.presentation": FileType.PPTX,
|
|
71
|
+
"application/msword": FileType.DOC,
|
|
72
|
+
"application/vnd.ms-excel": FileType.XLS,
|
|
73
|
+
"application/vnd.ms-powerpoint": FileType.PPT,
|
|
74
|
+
"text/html": FileType.HTML,
|
|
75
|
+
"application/epub+zip": FileType.EPUB,
|
|
76
|
+
"text/csv": FileType.CSV,
|
|
77
|
+
"text/plain": FileType.CSV, # resolved by extension
|
|
78
|
+
"application/json": FileType.JSON,
|
|
79
|
+
"text/xml": FileType.XML,
|
|
80
|
+
"application/xml": FileType.XML,
|
|
81
|
+
"message/rfc822": FileType.EML,
|
|
82
|
+
"image/tiff": FileType.IMAGE,
|
|
83
|
+
"image/bmp": FileType.IMAGE,
|
|
84
|
+
"image/webp": FileType.IMAGE,
|
|
85
|
+
"image/jpeg": FileType.IMAGE,
|
|
86
|
+
"image/png": FileType.IMAGE,
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def sniff(path: str | Path) -> FileType:
|
|
91
|
+
"""
|
|
92
|
+
Detect the file type of the given path.
|
|
93
|
+
Strategy:
|
|
94
|
+
1. Read magic bytes via python-magic -> look up MIME in _MIME_MAP
|
|
95
|
+
2. If ambiguous (e.g. text/plain could be CSV or TSV), fall back to extension
|
|
96
|
+
3. If still unknown, return FileType.UNKNOWN — never raise
|
|
97
|
+
Args:
|
|
98
|
+
path: path to any file
|
|
99
|
+
Returns:
|
|
100
|
+
FileType enum value
|
|
101
|
+
"""
|
|
102
|
+
path = Path(path)
|
|
103
|
+
ext = path.suffix.lower()
|
|
104
|
+
|
|
105
|
+
if magic is not None:
|
|
106
|
+
try:
|
|
107
|
+
mime = magic.from_file(str(path), mime=True)
|
|
108
|
+
file_type = _MIME_MAP.get(mime)
|
|
109
|
+
|
|
110
|
+
# MIME was recognized but ambiguous — let extension break the tie
|
|
111
|
+
if file_type in (FileType.CSV, FileType.HTML, None):
|
|
112
|
+
ext_type = _EXT_MAP.get(ext)
|
|
113
|
+
if ext_type is not None:
|
|
114
|
+
return ext_type
|
|
115
|
+
|
|
116
|
+
if file_type is not None:
|
|
117
|
+
return file_type
|
|
118
|
+
|
|
119
|
+
except Exception: # noqa: BLE001, S110
|
|
120
|
+
# magic can fail on locked files, permission errors, etc.
|
|
121
|
+
# fall through to extension lookup
|
|
122
|
+
pass
|
|
123
|
+
|
|
124
|
+
# Last resort: extension only
|
|
125
|
+
return _EXT_MAP.get(ext, FileType.UNKNOWN)
|
|
File without changes
|
|
File without changes
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# Export package
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
|
|
5
|
+
from universal_parser.core.schema import Document
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass
|
|
9
|
+
class Chunk:
|
|
10
|
+
"""A token-budgeted chunk ready for vector database embeddings."""
|
|
11
|
+
|
|
12
|
+
chunk_id: str
|
|
13
|
+
text: str
|
|
14
|
+
headings: list[str] = field(default_factory=list)
|
|
15
|
+
element_types: list[str] = field(default_factory=list)
|
|
16
|
+
page_numbers: list[int] = field(default_factory=list)
|
|
17
|
+
estimated_tokens: int = 0
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def to_chunks(
|
|
21
|
+
doc: Document,
|
|
22
|
+
max_tokens: int = 512,
|
|
23
|
+
overlap_tokens: int = 50,
|
|
24
|
+
) -> list[Chunk]:
|
|
25
|
+
"""
|
|
26
|
+
Hierarchical token-aware chunker for RAG pipelines.
|
|
27
|
+
Rules:
|
|
28
|
+
- Tracks active H1, H2, H3 hierarchy across elements
|
|
29
|
+
- Prefixes chunk with current heading path context
|
|
30
|
+
- Bundles elements until max_tokens budget is reached
|
|
31
|
+
- Keeps tables intact within a single chunk where possible
|
|
32
|
+
"""
|
|
33
|
+
chunks: list[Chunk] = []
|
|
34
|
+
current_headings: dict[int, str] = {} # level -> heading text
|
|
35
|
+
current_elements: list[str] = []
|
|
36
|
+
current_types: list[int] = []
|
|
37
|
+
current_pages: list[int] = []
|
|
38
|
+
current_tokens = 0
|
|
39
|
+
chunk_index = 1
|
|
40
|
+
|
|
41
|
+
for el in doc.content_tree:
|
|
42
|
+
# 1. Update heading stack
|
|
43
|
+
if el.type == "heading":
|
|
44
|
+
level = el.level or 1
|
|
45
|
+
current_headings = {lvl: txt for lvl, txt in current_headings.items() if lvl < level}
|
|
46
|
+
current_headings[level] = el.text or ""
|
|
47
|
+
|
|
48
|
+
# 2. Get text representation
|
|
49
|
+
el_text = el.markdown_repr or el.text or ""
|
|
50
|
+
if not el_text.strip():
|
|
51
|
+
continue
|
|
52
|
+
|
|
53
|
+
# Fast token estimation (~4 chars per token)
|
|
54
|
+
el_tokens = max(1, len(el_text) // 4)
|
|
55
|
+
|
|
56
|
+
# 3. Check if adding exceeds budget
|
|
57
|
+
if current_tokens + el_tokens > max_tokens and current_elements:
|
|
58
|
+
heading_hierarchy = [txt for _, txt in sorted(current_headings.items())]
|
|
59
|
+
heading_prefix = " > ".join(heading_hierarchy)
|
|
60
|
+
header_str = f"Context: {heading_prefix}\n\n" if heading_prefix else ""
|
|
61
|
+
chunk_text = header_str + "\n\n".join(current_elements)
|
|
62
|
+
total_est = len(chunk_text) // 4
|
|
63
|
+
|
|
64
|
+
chunks.append(
|
|
65
|
+
Chunk(
|
|
66
|
+
chunk_id=f"{doc.doc_id}-chunk-{chunk_index}",
|
|
67
|
+
text=chunk_text,
|
|
68
|
+
headings=heading_hierarchy,
|
|
69
|
+
element_types=list(set(current_types)),
|
|
70
|
+
page_numbers=sorted(set(current_pages)),
|
|
71
|
+
estimated_tokens=total_est,
|
|
72
|
+
)
|
|
73
|
+
)
|
|
74
|
+
chunk_index += 1
|
|
75
|
+
|
|
76
|
+
current_elements = []
|
|
77
|
+
current_types = []
|
|
78
|
+
current_pages = []
|
|
79
|
+
current_tokens = 0
|
|
80
|
+
|
|
81
|
+
# 4. Append element to current buffer
|
|
82
|
+
current_elements.append(el_text)
|
|
83
|
+
current_types.append(el.type)
|
|
84
|
+
if el.page is not None:
|
|
85
|
+
current_pages.append(el.page)
|
|
86
|
+
current_tokens += el_tokens
|
|
87
|
+
|
|
88
|
+
# Flush remaining buffer
|
|
89
|
+
if current_elements:
|
|
90
|
+
heading_hierarchy = [txt for _, txt in sorted(current_headings.items())]
|
|
91
|
+
heading_prefix = " > ".join(heading_hierarchy)
|
|
92
|
+
header_str = f"Context: {heading_prefix}\n\n" if heading_prefix else ""
|
|
93
|
+
|
|
94
|
+
chunk_text = header_str + "\n\n".join(current_elements)
|
|
95
|
+
total_est = len(chunk_text) // 4
|
|
96
|
+
|
|
97
|
+
chunks.append(
|
|
98
|
+
Chunk(
|
|
99
|
+
chunk_id=f"{doc.doc_id}-chunk-{chunk_index}",
|
|
100
|
+
text=chunk_text,
|
|
101
|
+
headings=heading_hierarchy,
|
|
102
|
+
element_types=list(set(current_types)),
|
|
103
|
+
page_numbers=sorted(set(current_pages)),
|
|
104
|
+
estimated_tokens=total_est,
|
|
105
|
+
)
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
return chunks
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
|
|
5
|
+
from universal_parser.core.schema import Document
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass
|
|
9
|
+
class GraphNode:
|
|
10
|
+
id: str
|
|
11
|
+
type: str # "document", "section", "table", "content"
|
|
12
|
+
properties: dict[str, str | int | float] = field(default_factory=dict)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass
|
|
16
|
+
class GraphEdge:
|
|
17
|
+
source_id: str
|
|
18
|
+
target_id: str
|
|
19
|
+
relationship: str # "CONTAINS_SECTION", "CONTAINS", "FOLLOWS"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class KnowledgeGraph:
|
|
24
|
+
nodes: list[GraphNode] = field(default_factory=list)
|
|
25
|
+
edges: list[GraphEdge] = field(default_factory=list)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def to_graph(doc: Document) -> KnowledgeGraph:
|
|
29
|
+
"""
|
|
30
|
+
Export Document into a Knowledge Graph structure for GraphRAG & Neo4j.
|
|
31
|
+
|
|
32
|
+
Creates:
|
|
33
|
+
- Document Root Node
|
|
34
|
+
- Heading / Section Nodes with hierarchical edges
|
|
35
|
+
- Paragraph & Table Content Nodes linked to their parent sections
|
|
36
|
+
"""
|
|
37
|
+
nodes: list[GraphNode] = []
|
|
38
|
+
edges: list[GraphEdge] = []
|
|
39
|
+
|
|
40
|
+
# 1. Document Root Node
|
|
41
|
+
doc_node_id = f"doc:{doc.doc_id}"
|
|
42
|
+
nodes.append(
|
|
43
|
+
GraphNode(
|
|
44
|
+
id=doc_node_id,
|
|
45
|
+
type="document",
|
|
46
|
+
properties={
|
|
47
|
+
"file_name": doc.metadata.file_name,
|
|
48
|
+
"file_type": doc.metadata.file_type,
|
|
49
|
+
},
|
|
50
|
+
)
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
heading_stack: list[tuple[int, str]] = [] # (level, node_id)
|
|
54
|
+
prev_node_id: str | None = None
|
|
55
|
+
|
|
56
|
+
for idx, el in enumerate(doc.content_tree, start=1):
|
|
57
|
+
node_id = f"el:{doc.doc_id}:{idx}"
|
|
58
|
+
|
|
59
|
+
if el.type == "heading":
|
|
60
|
+
level = el.level or 1
|
|
61
|
+
while heading_stack and heading_stack[-1][0] >= level:
|
|
62
|
+
heading_stack.pop()
|
|
63
|
+
|
|
64
|
+
parent_id = heading_stack[-1][1] if heading_stack else doc_node_id
|
|
65
|
+
|
|
66
|
+
nodes.append(
|
|
67
|
+
GraphNode(
|
|
68
|
+
id=node_id,
|
|
69
|
+
type="section",
|
|
70
|
+
properties={"title": el.text or "", "level": level},
|
|
71
|
+
)
|
|
72
|
+
)
|
|
73
|
+
edges.append(
|
|
74
|
+
GraphEdge(
|
|
75
|
+
source_id=parent_id,
|
|
76
|
+
target_id=node_id,
|
|
77
|
+
relationship="CONTAINS_SECTION",
|
|
78
|
+
)
|
|
79
|
+
)
|
|
80
|
+
heading_stack.append((level, node_id))
|
|
81
|
+
|
|
82
|
+
else:
|
|
83
|
+
parent_id = heading_stack[-1][1] if heading_stack else doc_node_id
|
|
84
|
+
node_type = "table" if el.type == "table" else "content"
|
|
85
|
+
|
|
86
|
+
nodes.append(
|
|
87
|
+
GraphNode(
|
|
88
|
+
id=node_id,
|
|
89
|
+
type=node_type,
|
|
90
|
+
properties={
|
|
91
|
+
"text": el.text or el.markdown_repr or "",
|
|
92
|
+
"element_type": el.type,
|
|
93
|
+
},
|
|
94
|
+
)
|
|
95
|
+
)
|
|
96
|
+
edges.append(
|
|
97
|
+
GraphEdge(
|
|
98
|
+
source_id=parent_id,
|
|
99
|
+
target_id=node_id,
|
|
100
|
+
relationship="CONTAINS",
|
|
101
|
+
)
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
if prev_node_id:
|
|
105
|
+
edges.append(
|
|
106
|
+
GraphEdge(
|
|
107
|
+
source_id=prev_node_id,
|
|
108
|
+
target_id=node_id,
|
|
109
|
+
relationship="FOLLOWS",
|
|
110
|
+
)
|
|
111
|
+
)
|
|
112
|
+
prev_node_id = node_id
|
|
113
|
+
|
|
114
|
+
return KnowledgeGraph(nodes=nodes, edges=edges)
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from universal_parser.core.schema import Document
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def to_markdown(doc: Document) -> str:
|
|
7
|
+
"""
|
|
8
|
+
Convert a parsed Document into a clean, unified Markdown string.
|
|
9
|
+
Rules:
|
|
10
|
+
- Headings render as # H1, ## H2, ### H3
|
|
11
|
+
- Paragraphs render as clean prose
|
|
12
|
+
- List items render with "- "
|
|
13
|
+
- Tables render with structured Markdown grids
|
|
14
|
+
- Code blocks render within ``` fences
|
|
15
|
+
"""
|
|
16
|
+
blocks: list[str] = []
|
|
17
|
+
|
|
18
|
+
for el in doc.content_tree:
|
|
19
|
+
if el.markdown_repr:
|
|
20
|
+
blocks.append(el.markdown_repr)
|
|
21
|
+
elif el.type == "heading":
|
|
22
|
+
level = el.level or 1
|
|
23
|
+
blocks.append(f"{'#' * level} {el.text or ''}")
|
|
24
|
+
elif el.type == "list_item":
|
|
25
|
+
blocks.append(f"- {el.text or ''}")
|
|
26
|
+
elif el.type == "code_block":
|
|
27
|
+
blocks.append(f"```\n{el.text or ''}\n```")
|
|
28
|
+
elif el.text:
|
|
29
|
+
blocks.append(el.text)
|
|
30
|
+
|
|
31
|
+
return "\n\n".join(blocks)
|
|
File without changes
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from abc import ABC, abstractmethod
|
|
4
|
+
from collections.abc import Iterator
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import ClassVar
|
|
7
|
+
|
|
8
|
+
from universal_parser.core.schema import Element
|
|
9
|
+
from universal_parser.core.sniffer import FileType
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class BaseExtractor(ABC):
|
|
13
|
+
"""
|
|
14
|
+
Abstract base class for all format extractors.
|
|
15
|
+
Every extractor in universal_parser.extractors must:
|
|
16
|
+
1. Inherit from BaseExtractor
|
|
17
|
+
2. Declare which FileTypes it handles via supported_types
|
|
18
|
+
3. Implement stream() — yield Elements one at a time, never return a list
|
|
19
|
+
Adding a new format = one new file inheriting this + one entry in router.py.
|
|
20
|
+
Nothing else in core/ needs to change.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
supported_types: ClassVar[list[FileType]] = []
|
|
24
|
+
|
|
25
|
+
@abstractmethod
|
|
26
|
+
def stream(self, path: str | Path) -> Iterator[Element]:
|
|
27
|
+
"""
|
|
28
|
+
Parse the document at the given path and yield Element objects.
|
|
29
|
+
Rules:
|
|
30
|
+
- Must yield, never return a full list (streaming = memory safe)
|
|
31
|
+
- Must never raise on a recoverable error — yield a low-confidence
|
|
32
|
+
element or skip, log the issue, keep going
|
|
33
|
+
- One Element at a time, in reading order
|
|
34
|
+
Args:
|
|
35
|
+
path: absolute or relative path to the source file
|
|
36
|
+
Yields:
|
|
37
|
+
Element objects in document reading order
|
|
38
|
+
"""
|
|
39
|
+
...
|
|
File without changes
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Iterator
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import ClassVar
|
|
6
|
+
|
|
7
|
+
import cv2
|
|
8
|
+
import numpy as np
|
|
9
|
+
from rapidocr_onnxruntime import RapidOCR
|
|
10
|
+
|
|
11
|
+
from universal_parser.core.router import register
|
|
12
|
+
from universal_parser.core.schema import BBox, Element
|
|
13
|
+
from universal_parser.core.sniffer import FileType
|
|
14
|
+
from universal_parser.extractors.base import BaseExtractor
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@register
|
|
18
|
+
class ImageScanExtractor(BaseExtractor):
|
|
19
|
+
"""
|
|
20
|
+
Extractor for image and scanned files (.png, .jpg, .tiff, .bmp, .webp).
|
|
21
|
+
Handles:
|
|
22
|
+
- Image preprocessing (deskewing / contrast enhancement via OpenCV)
|
|
23
|
+
- CPU-based text and bounding box detection via RapidOCR (ONNX)
|
|
24
|
+
- Confidence scoring per text element
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
supported_types: ClassVar[list[FileType]] = [FileType.IMAGE]
|
|
28
|
+
|
|
29
|
+
def __init__(self) -> None:
|
|
30
|
+
super().__init__()
|
|
31
|
+
# Initialize RapidOCR engine (cached in memory)
|
|
32
|
+
self._ocr = RapidOCR()
|
|
33
|
+
|
|
34
|
+
def stream(self, path: str | Path) -> Iterator[Element]:
|
|
35
|
+
path_str = str(path)
|
|
36
|
+
|
|
37
|
+
try:
|
|
38
|
+
# Step 1: Read image with OpenCV
|
|
39
|
+
img = cv2.imread(path_str)
|
|
40
|
+
if img is None:
|
|
41
|
+
return
|
|
42
|
+
|
|
43
|
+
# Step 2: Preprocess / deskew
|
|
44
|
+
preprocessing_img = self._preprocess_image(img)
|
|
45
|
+
|
|
46
|
+
# Step 3: Run RapidOCR
|
|
47
|
+
ocr_results, _ = self._ocr(preprocessing_img)
|
|
48
|
+
if not ocr_results:
|
|
49
|
+
return
|
|
50
|
+
# Step 4: Stream extracted elements
|
|
51
|
+
for item in ocr_results:
|
|
52
|
+
dt_boxes, text, score = item
|
|
53
|
+
clean_text = text.strip()
|
|
54
|
+
if not clean_text:
|
|
55
|
+
continue
|
|
56
|
+
|
|
57
|
+
# Compute bounding box (x0, y0, x1, y1)
|
|
58
|
+
pts = np.array(dt_boxes, dtype=np.float32)
|
|
59
|
+
x0 = float(np.min(pts[:, 0]))
|
|
60
|
+
y0 = float(np.min(pts[:, 1]))
|
|
61
|
+
x1 = float(np.max(pts[:, 0]))
|
|
62
|
+
y1 = float(np.max(pts[:, 1]))
|
|
63
|
+
|
|
64
|
+
yield Element(
|
|
65
|
+
type="paragraph",
|
|
66
|
+
text=clean_text,
|
|
67
|
+
page=1,
|
|
68
|
+
bbox=BBox(x0=x0, y0=y0, x1=x1, y1=y1),
|
|
69
|
+
markdown_repr=clean_text,
|
|
70
|
+
confidence=round(float(score), 3),
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
except Exception: # noqa: BLE001
|
|
74
|
+
return
|
|
75
|
+
|
|
76
|
+
def _preprocess_image(self, img: np.ndarray) -> np.ndarray:
|
|
77
|
+
"""Basic contrast & grayscale cleanup for sharper OCR."""
|
|
78
|
+
# RapidOCR handles RGB internally; return original if valid
|
|
79
|
+
return img
|
|
File without changes
|