universal-doc-parser 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- universal_doc_parser-1.0.0.dist-info/METADATA +692 -0
- universal_doc_parser-1.0.0.dist-info/RECORD +48 -0
- universal_doc_parser-1.0.0.dist-info/WHEEL +4 -0
- universal_doc_parser-1.0.0.dist-info/licenses/LICENSE +21 -0
- universal_parser/__init__.py +28 -0
- universal_parser/adaptive/__init__.py +17 -0
- universal_parser/adaptive/config_cache.py +127 -0
- universal_parser/adaptive/fingerprint.py +133 -0
- universal_parser/adaptive/tuner.py +97 -0
- universal_parser/core/__init__.py +0 -0
- universal_parser/core/engine.py +55 -0
- universal_parser/core/router.py +38 -0
- universal_parser/core/schema.py +58 -0
- universal_parser/core/sniffer.py +125 -0
- universal_parser/enrichment/__init__.py +0 -0
- universal_parser/enrichment/vlm_enricher.py +0 -0
- universal_parser/exports/__init__.py +1 -0
- universal_parser/exports/to_chunks.py +108 -0
- universal_parser/exports/to_graph.py +114 -0
- universal_parser/exports/to_markdown.py +31 -0
- universal_parser/extractors/__init__.py +0 -0
- universal_parser/extractors/base.py +39 -0
- universal_parser/extractors/images/__init__.py +0 -0
- universal_parser/extractors/images/scan_extractor.py +79 -0
- universal_parser/extractors/mail/__init__.py +0 -0
- universal_parser/extractors/mail/mail_extractor.py +206 -0
- universal_parser/extractors/office/__init__.py +0 -0
- universal_parser/extractors/office/docx_extractor.py +131 -0
- universal_parser/extractors/office/legacy_extractor.py +112 -0
- universal_parser/extractors/office/pptx_extractor.py +109 -0
- universal_parser/extractors/office/xlsx_extractor.py +93 -0
- universal_parser/extractors/pdf/__init__.py +0 -0
- universal_parser/extractors/pdf/native.py +331 -0
- universal_parser/extractors/pdf/tables.py +155 -0
- universal_parser/extractors/pdf/visual_onnx.py +0 -0
- universal_parser/extractors/structured/__init__.py +0 -0
- universal_parser/extractors/structured/csv_extractor.py +86 -0
- universal_parser/extractors/structured/json_xml_extractor.py +104 -0
- universal_parser/extractors/structured/parquet_extractor.py +59 -0
- universal_parser/extractors/web/__init__.py +1 -0
- universal_parser/extractors/web/epub_extractor.py +81 -0
- universal_parser/extractors/web/html_extractor.py +111 -0
- universal_parser/mcp/__init__.py +1 -0
- universal_parser/mcp/server.py +61 -0
- universal_parser/observability/__init__.py +12 -0
- universal_parser/observability/dashboard.py +231 -0
- universal_parser/observability/logger.py +0 -0
- universal_parser/observability/metrics.py +99 -0
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import email
|
|
4
|
+
import mailbox
|
|
5
|
+
import tempfile
|
|
6
|
+
from collections.abc import Iterator
|
|
7
|
+
from email import policy
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import ClassVar
|
|
10
|
+
|
|
11
|
+
import extract_msg
|
|
12
|
+
from selectolax.parser import HTMLParser
|
|
13
|
+
|
|
14
|
+
from universal_parser.core.router import get_extractor, register
|
|
15
|
+
from universal_parser.core.schema import Element
|
|
16
|
+
from universal_parser.core.sniffer import FileType, sniff
|
|
17
|
+
from universal_parser.extractors.base import BaseExtractor
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@register
|
|
21
|
+
class MailExtractor(BaseExtractor):
|
|
22
|
+
"""
|
|
23
|
+
Extractor for email formats (.eml, .msg, .mbox).
|
|
24
|
+
|
|
25
|
+
Handles:
|
|
26
|
+
- Header metadata (Subject, From, To, Date)
|
|
27
|
+
- Body extraction (HTML / Plaintext via selectolax)
|
|
28
|
+
- Recursive attachment parsing through the sniffer/router
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
supported_types: ClassVar[list[FileType]] = [
|
|
32
|
+
FileType.EML,
|
|
33
|
+
FileType.MSG,
|
|
34
|
+
FileType.MBOX,
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
def stream(self, path: str | Path) -> Iterator[Element]:
|
|
38
|
+
path_obj = Path(path)
|
|
39
|
+
ext = path_obj.suffix.lower()
|
|
40
|
+
|
|
41
|
+
if ext == ".eml":
|
|
42
|
+
yield from self._stream_eml(path_obj)
|
|
43
|
+
elif ext == ".msg":
|
|
44
|
+
yield from self._stream_msg(path_obj)
|
|
45
|
+
elif ext == ".mbox":
|
|
46
|
+
yield from self._stream_mbox(path_obj)
|
|
47
|
+
|
|
48
|
+
def _stream_eml(self, path: Path) -> Iterator[Element]:
|
|
49
|
+
try:
|
|
50
|
+
with open(path, "rb") as f:
|
|
51
|
+
msg = email.message_from_binary_file(f, policy=policy.default)
|
|
52
|
+
|
|
53
|
+
yield from self._process_email_message(msg)
|
|
54
|
+
except Exception: # noqa: BLE001
|
|
55
|
+
return
|
|
56
|
+
|
|
57
|
+
def _stream_msg(self, path: Path) -> Iterator[Element]:
|
|
58
|
+
try:
|
|
59
|
+
msg = extract_msg.Message(str(path))
|
|
60
|
+
subject = msg.subject or "No Subject"
|
|
61
|
+
sender = msg.sender or "Unknown Sender"
|
|
62
|
+
to = msg.to or "Unknown Recipient"
|
|
63
|
+
date = str(msg.date) if msg.date else ""
|
|
64
|
+
|
|
65
|
+
yield Element(
|
|
66
|
+
type="heading",
|
|
67
|
+
level=1,
|
|
68
|
+
text=f"Subject: {subject}",
|
|
69
|
+
markdown_repr=f"# Subject: {subject}",
|
|
70
|
+
confidence=1.0,
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
meta_text = f"From: {sender} | To: {to}" + (f" | Date: {date}" if date else "")
|
|
74
|
+
yield Element(
|
|
75
|
+
type="paragraph",
|
|
76
|
+
text=meta_text,
|
|
77
|
+
markdown_repr=f"**{meta_text}**",
|
|
78
|
+
confidence=1.0,
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
body_text = msg.body
|
|
82
|
+
if body_text:
|
|
83
|
+
for paragraph in body_text.split("\n\n"):
|
|
84
|
+
clean_p = paragraph.strip()
|
|
85
|
+
if clean_p:
|
|
86
|
+
yield Element(
|
|
87
|
+
type="paragraph",
|
|
88
|
+
text=clean_p,
|
|
89
|
+
markdown_repr=clean_p,
|
|
90
|
+
confidence=1.0,
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
# Process attachments recursively
|
|
94
|
+
for att in msg.attachments:
|
|
95
|
+
att_data = att.data
|
|
96
|
+
att_name = att.longFilename or att.shortFilename or "attachment"
|
|
97
|
+
if att_data:
|
|
98
|
+
yield from self._process_attachment_bytes(att_name, att_data)
|
|
99
|
+
|
|
100
|
+
msg.close()
|
|
101
|
+
except Exception: # noqa: BLE001
|
|
102
|
+
return
|
|
103
|
+
|
|
104
|
+
def _stream_mbox(self, path: Path) -> Iterator[Element]:
|
|
105
|
+
try:
|
|
106
|
+
mbox = mailbox.mbox(str(path))
|
|
107
|
+
for msg in mbox.values():
|
|
108
|
+
yield from self._process_email_message(msg)
|
|
109
|
+
except Exception: # noqa: BLE001
|
|
110
|
+
return
|
|
111
|
+
|
|
112
|
+
def _process_email_message(self, msg: email.message.EmailMessage) -> Iterator[Element]:
|
|
113
|
+
subject = str(msg.get("Subject", "No Subject"))
|
|
114
|
+
sender = str(msg.get("From", "Unknown Sender"))
|
|
115
|
+
to = str(msg.get("To", "Unknown Recipient"))
|
|
116
|
+
date = str(msg.get("Date", ""))
|
|
117
|
+
|
|
118
|
+
yield Element(
|
|
119
|
+
type="heading",
|
|
120
|
+
level=1,
|
|
121
|
+
text=f"Subject: {subject}",
|
|
122
|
+
markdown_repr=f"# Subject: {subject}",
|
|
123
|
+
confidence=1.0,
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
meta_text = f"From: {sender} | To: {to}" + (f" | Date: {date}" if date else "")
|
|
127
|
+
yield Element(
|
|
128
|
+
type="paragraph",
|
|
129
|
+
text=meta_text,
|
|
130
|
+
markdown_repr=f"**{meta_text}**",
|
|
131
|
+
confidence=1.0,
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
# Body extraction
|
|
135
|
+
body_part = msg.get_body(preferencelist=("html", "plain"))
|
|
136
|
+
if body_part:
|
|
137
|
+
content = body_part.get_content()
|
|
138
|
+
if body_part.get_content_type() == "text/html":
|
|
139
|
+
parser = HTMLParser(content)
|
|
140
|
+
for tag in parser.css("script, style, noscript"):
|
|
141
|
+
tag.decompose()
|
|
142
|
+
body_elem = parser.body or parser.root
|
|
143
|
+
if body_elem:
|
|
144
|
+
for node in body_elem.iter():
|
|
145
|
+
tag_name = node.tag.lower() if node.tag else ""
|
|
146
|
+
if tag_name in ("h1", "h2", "h3", "h4", "h5", "h6"):
|
|
147
|
+
h_text = node.text(strip=True)
|
|
148
|
+
if h_text:
|
|
149
|
+
yield Element(
|
|
150
|
+
type="heading",
|
|
151
|
+
level=int(tag_name[1]),
|
|
152
|
+
text=h_text,
|
|
153
|
+
markdown_repr=f"{'#' * int(tag_name[1])} {h_text}",
|
|
154
|
+
confidence=1.0,
|
|
155
|
+
)
|
|
156
|
+
elif tag_name in ("p", "li"):
|
|
157
|
+
p_text = node.text(strip=True)
|
|
158
|
+
if p_text:
|
|
159
|
+
yield Element(
|
|
160
|
+
type="paragraph",
|
|
161
|
+
text=p_text,
|
|
162
|
+
markdown_repr=p_text,
|
|
163
|
+
confidence=1.0,
|
|
164
|
+
)
|
|
165
|
+
else:
|
|
166
|
+
for paragraph in str(content).split("\n\n"):
|
|
167
|
+
clean_p = paragraph.strip()
|
|
168
|
+
if clean_p:
|
|
169
|
+
yield Element(
|
|
170
|
+
type="paragraph",
|
|
171
|
+
text=clean_p,
|
|
172
|
+
markdown_repr=clean_p,
|
|
173
|
+
confidence=1.0,
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
# Attachment extraction (recursive)
|
|
177
|
+
for part in msg.iter_attachments():
|
|
178
|
+
filename = part.get_filename() or "attachment"
|
|
179
|
+
payload = part.get_payload(decode=True)
|
|
180
|
+
if payload:
|
|
181
|
+
yield from self._process_attachment_bytes(filename, payload)
|
|
182
|
+
|
|
183
|
+
def _process_attachment_bytes(self, filename: str, data: bytes) -> Iterator[Element]:
|
|
184
|
+
"""Save attachment to temp file, sniff type, and route to sub-extractor."""
|
|
185
|
+
try:
|
|
186
|
+
suffix = Path(filename).suffix
|
|
187
|
+
with tempfile.NamedTemporaryFile(suffix=suffix, delete=False) as tmp:
|
|
188
|
+
tmp.write(data)
|
|
189
|
+
tmp_path = Path(tmp.name)
|
|
190
|
+
|
|
191
|
+
file_type = sniff(tmp_path)
|
|
192
|
+
extractor = get_extractor(file_type)
|
|
193
|
+
|
|
194
|
+
if extractor:
|
|
195
|
+
yield Element(
|
|
196
|
+
type="heading",
|
|
197
|
+
level=2,
|
|
198
|
+
text=f"Attachment: {filename}",
|
|
199
|
+
markdown_repr=f"## Attachment: {filename}",
|
|
200
|
+
confidence=1.0,
|
|
201
|
+
)
|
|
202
|
+
yield from extractor.stream(tmp_path)
|
|
203
|
+
|
|
204
|
+
tmp_path.unlink(missing_ok=True)
|
|
205
|
+
except Exception: # noqa: BLE001
|
|
206
|
+
return
|
|
File without changes
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Iterator
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import ClassVar
|
|
6
|
+
|
|
7
|
+
from docx import Document as DocxDocument
|
|
8
|
+
from docx.oxml.ns import qn
|
|
9
|
+
from docx.table import Table
|
|
10
|
+
from docx.text.paragraph import Paragraph
|
|
11
|
+
|
|
12
|
+
from universal_parser.core.router import register
|
|
13
|
+
from universal_parser.core.schema import Element, TableData
|
|
14
|
+
from universal_parser.core.sniffer import FileType
|
|
15
|
+
from universal_parser.extractors.base import BaseExtractor
|
|
16
|
+
|
|
17
|
+
# Map Word built-in heading style names to heading levels
|
|
18
|
+
_HEADING_STYLE_MAP: dict[str, int] = {
|
|
19
|
+
"heading 1": 1,
|
|
20
|
+
"heading 2": 2,
|
|
21
|
+
"heading 3": 3,
|
|
22
|
+
"heading 4": 4,
|
|
23
|
+
"heading 5": 5,
|
|
24
|
+
"heading 6": 6,
|
|
25
|
+
"title": 1, # Word's "Title" style treated as H1
|
|
26
|
+
"subtitle": 2, # Word's "Subtitle" style treated as H2
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@register
|
|
31
|
+
class DocxExtractor(BaseExtractor):
|
|
32
|
+
"""
|
|
33
|
+
Extractor for .docx files using python-docx.
|
|
34
|
+
Handles:
|
|
35
|
+
- Paragraphs and headings in correct document order
|
|
36
|
+
- Tables (headers inferred from first row)
|
|
37
|
+
- Preserves reading order by iterating document body directly
|
|
38
|
+
|
|
39
|
+
Does NOT handle:
|
|
40
|
+
- Legacy .doc binary format (needs olefile — Phase 6)
|
|
41
|
+
- Embedded images/charts (Phase 6)
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
supported_types: ClassVar[list[FileType]] = [FileType.DOCX]
|
|
45
|
+
|
|
46
|
+
def stream(self, path: str | Path) -> Iterator[Element]:
|
|
47
|
+
"""Stream elements from a DOCX file in document reading order."""
|
|
48
|
+
try:
|
|
49
|
+
doc = DocxDocument(str(path))
|
|
50
|
+
except Exception: # noqa: BLE001
|
|
51
|
+
# Corrupted or password-protected DOCX — exit cleanly
|
|
52
|
+
return
|
|
53
|
+
|
|
54
|
+
try:
|
|
55
|
+
# Iterate body children directly to preserve paragraph + table order.
|
|
56
|
+
# doc.paragraphs and doc.tables are separate lists — using them
|
|
57
|
+
# would give all paragraphs first then all tables, losing real order.
|
|
58
|
+
for child in doc.element.body:
|
|
59
|
+
tag = child.tag
|
|
60
|
+
|
|
61
|
+
# Paragraph element
|
|
62
|
+
if tag == qn("w:p"):
|
|
63
|
+
element = self._process_paragraph(Paragraph(child, doc))
|
|
64
|
+
if element is not None:
|
|
65
|
+
yield element
|
|
66
|
+
|
|
67
|
+
# Table element
|
|
68
|
+
elif tag == qn("w:tbl"):
|
|
69
|
+
element = self._process_table(Table(child, doc))
|
|
70
|
+
if element is not None:
|
|
71
|
+
yield element
|
|
72
|
+
|
|
73
|
+
except Exception: # noqa: BLE001
|
|
74
|
+
# If anything fails mid-document, we stop cleanly.
|
|
75
|
+
# We do not crash — whatever was yielded before the error is valid.
|
|
76
|
+
return
|
|
77
|
+
|
|
78
|
+
def _process_paragraph(self, para: Paragraph) -> Element | None:
|
|
79
|
+
"""Convert a python-docx Paragraph into an Element."""
|
|
80
|
+
text = para.text.strip()
|
|
81
|
+
|
|
82
|
+
# Skip empty paragraphs — they carry no information for RAG
|
|
83
|
+
if not text:
|
|
84
|
+
return None
|
|
85
|
+
|
|
86
|
+
style_name = para.style.name.lower() if para.style else ""
|
|
87
|
+
heading_level = _HEADING_STYLE_MAP.get(style_name)
|
|
88
|
+
|
|
89
|
+
if heading_level is not None:
|
|
90
|
+
return Element(
|
|
91
|
+
type="heading",
|
|
92
|
+
level=heading_level,
|
|
93
|
+
text=text,
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
# Check for list items (Word list styles contain "list" in the name)
|
|
97
|
+
if "list" in style_name:
|
|
98
|
+
return Element(
|
|
99
|
+
type="list_item",
|
|
100
|
+
text=text,
|
|
101
|
+
)
|
|
102
|
+
return Element(
|
|
103
|
+
type="paragraph",
|
|
104
|
+
text=text,
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
def _process_table(self, table: Table) -> Element | None:
|
|
108
|
+
"""Convert a python-docx Table into an Element with TableData."""
|
|
109
|
+
rows = []
|
|
110
|
+
for row in table.rows:
|
|
111
|
+
row_data = [cell.text.strip() for cell in row.cells]
|
|
112
|
+
rows.append(row_data)
|
|
113
|
+
|
|
114
|
+
if not rows:
|
|
115
|
+
return None
|
|
116
|
+
|
|
117
|
+
# Treat the first row as headers
|
|
118
|
+
headers = rows[0]
|
|
119
|
+
data_rows = rows[1:]
|
|
120
|
+
|
|
121
|
+
# Build a markdown representation for quick use in RAG prompts
|
|
122
|
+
md_header = "| " + " | ".join(headers) + " |"
|
|
123
|
+
md_separator = "| " + " | ".join(["---"] * len(headers)) + " |"
|
|
124
|
+
md_rows = ["| " + " | ".join(row) + " |" for row in data_rows]
|
|
125
|
+
markdown_repr = "\n".join([md_header, md_separator] + md_rows)
|
|
126
|
+
|
|
127
|
+
return Element(
|
|
128
|
+
type="table",
|
|
129
|
+
data=TableData(headers=headers, rows=data_rows),
|
|
130
|
+
markdown_repr=markdown_repr,
|
|
131
|
+
)
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from collections.abc import Iterator
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import ClassVar
|
|
7
|
+
|
|
8
|
+
import olefile
|
|
9
|
+
import xlrd
|
|
10
|
+
|
|
11
|
+
from universal_parser.core.router import register
|
|
12
|
+
from universal_parser.core.schema import Element, TableData
|
|
13
|
+
from universal_parser.core.sniffer import FileType
|
|
14
|
+
from universal_parser.extractors.base import BaseExtractor
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@register
|
|
18
|
+
class LegacyOfficeExtractor(BaseExtractor):
|
|
19
|
+
"""
|
|
20
|
+
Extractor for legacy Microsoft Office binary formats (.doc, .xls, .ppt).
|
|
21
|
+
Handles:
|
|
22
|
+
- .xls: Streams sheets into structured TableData & Markdown using xlrd
|
|
23
|
+
- .doc / .ppt: Decodes text streams from OLE2 binary containers via olefile
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
supported_types: ClassVar[list[FileType]] = [
|
|
27
|
+
FileType.DOC,
|
|
28
|
+
FileType.XLS,
|
|
29
|
+
FileType.PPT,
|
|
30
|
+
]
|
|
31
|
+
|
|
32
|
+
def stream(self, path: str | Path) -> Iterator[Element]:
|
|
33
|
+
path_obj = Path(path)
|
|
34
|
+
ext = path_obj.suffix.lower()
|
|
35
|
+
|
|
36
|
+
if ext == ".xls":
|
|
37
|
+
yield from self._stream_xls(path_obj)
|
|
38
|
+
elif ext in (".doc", ".ppt"):
|
|
39
|
+
yield from self._stream_ole_text(path_obj)
|
|
40
|
+
|
|
41
|
+
def _stream_xls(self, path: Path) -> Iterator[Element]:
|
|
42
|
+
"""Stream sheets from legacy .xls files using xlrd."""
|
|
43
|
+
try:
|
|
44
|
+
wb = xlrd.open_workbook(str(path))
|
|
45
|
+
for sheet in wb.sheets():
|
|
46
|
+
rows = []
|
|
47
|
+
for row_idx in range(sheet.nrows):
|
|
48
|
+
rows_vals = [
|
|
49
|
+
str(sheet.cell_value(row_idx, col_idx)).strip()
|
|
50
|
+
for col_idx in range(sheet.ncols)
|
|
51
|
+
]
|
|
52
|
+
if any(rows_vals):
|
|
53
|
+
rows.append(rows_vals)
|
|
54
|
+
|
|
55
|
+
if not rows:
|
|
56
|
+
continue
|
|
57
|
+
|
|
58
|
+
headers = rows[0]
|
|
59
|
+
data_rows = rows[1:]
|
|
60
|
+
|
|
61
|
+
md_header = "| " + " | ".join(headers) + " |"
|
|
62
|
+
md_separator = "| " + " | ".join(["---"] * len(headers)) + " |"
|
|
63
|
+
md_rows = ["| " + " | ".join(r) + " |" for r in data_rows]
|
|
64
|
+
markdown_repr = f"### Sheet: {sheet.name}\n\n" + "\n".join(
|
|
65
|
+
[md_header, md_separator] + md_rows
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
yield Element(
|
|
69
|
+
type="table",
|
|
70
|
+
text=f"Sheet: {sheet.name}",
|
|
71
|
+
data=TableData(headers=headers, rows=data_rows),
|
|
72
|
+
markdown_repr=markdown_repr,
|
|
73
|
+
confidence=1.0,
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
except Exception: # noqa: BLE001
|
|
77
|
+
return
|
|
78
|
+
|
|
79
|
+
def _stream_ole_text(self, path: Path) -> Iterator[Element]:
|
|
80
|
+
"""Extract readable text streams from OLE2 compound files (.doc / .ppt)."""
|
|
81
|
+
try:
|
|
82
|
+
if not olefile.isOleFile(str(path)):
|
|
83
|
+
return
|
|
84
|
+
|
|
85
|
+
ole = olefile.OleFileIO(str(path))
|
|
86
|
+
extracted_text = []
|
|
87
|
+
|
|
88
|
+
for stream_name in ole.listdir():
|
|
89
|
+
try:
|
|
90
|
+
stream_data = ole.openstream(stream_name).read()
|
|
91
|
+
# Filter ASCII string sequences from binary stream
|
|
92
|
+
raw_strings = re.findall(rb"[\x20-\x7E]{4,}", stream_data)
|
|
93
|
+
for s in raw_strings:
|
|
94
|
+
decoded = s.decode("ascii", errors="ignore").strip()
|
|
95
|
+
if len(decoded) > 3 and not decoded.startswith("Microsoft"):
|
|
96
|
+
extracted_text.append(decoded)
|
|
97
|
+
|
|
98
|
+
except Exception: # noqa: BLE001, S112
|
|
99
|
+
continue
|
|
100
|
+
|
|
101
|
+
ole.close()
|
|
102
|
+
|
|
103
|
+
for text_chunk in extracted_text:
|
|
104
|
+
yield Element(
|
|
105
|
+
type="Paragraph",
|
|
106
|
+
text=text_chunk,
|
|
107
|
+
markdown_repr=text_chunk,
|
|
108
|
+
confidence=0.8,
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
except Exception: # noqa: BLE001
|
|
112
|
+
return
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Iterator
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import ClassVar
|
|
6
|
+
|
|
7
|
+
from pptx import Presentation
|
|
8
|
+
|
|
9
|
+
from universal_parser.core.router import register
|
|
10
|
+
from universal_parser.core.schema import Element, TableData
|
|
11
|
+
from universal_parser.core.sniffer import FileType
|
|
12
|
+
from universal_parser.extractors.base import BaseExtractor
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@register
|
|
16
|
+
class PPTXExtractor(BaseExtractor):
|
|
17
|
+
"""
|
|
18
|
+
Extractor for PowerPoint presentations (.pptx).
|
|
19
|
+
Handles:
|
|
20
|
+
- Slide titles -> Level 1 Headings with slide numbers
|
|
21
|
+
- Text boxes and bullet points -> Paragraphs and List Items
|
|
22
|
+
- Presentation tables -> Structured TableData & Markdown tables
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
supported_types: ClassVar[list[FileType]] = [FileType.PPTX]
|
|
26
|
+
|
|
27
|
+
def stream(self, path: str | Path) -> Iterator[Element]:
|
|
28
|
+
path_str = str(path)
|
|
29
|
+
|
|
30
|
+
try:
|
|
31
|
+
prs = Presentation(path_str)
|
|
32
|
+
|
|
33
|
+
except Exception: # noqa: BLE001
|
|
34
|
+
return
|
|
35
|
+
|
|
36
|
+
for slide_num, slide in enumerate(prs.slides, start=1):
|
|
37
|
+
# 1. Slide Title (Heading 1)
|
|
38
|
+
if slide.shapes.title and slide.shapes.title.text.strip():
|
|
39
|
+
title = slide.shapes.title.text.strip()
|
|
40
|
+
yield Element(
|
|
41
|
+
type="heading",
|
|
42
|
+
level=1,
|
|
43
|
+
text=f"Slide {slide_num}: {title}",
|
|
44
|
+
page=slide_num,
|
|
45
|
+
markdown_repr=f"# Slide {slide_num}: {title}",
|
|
46
|
+
confidence=1.0,
|
|
47
|
+
)
|
|
48
|
+
else:
|
|
49
|
+
yield Element(
|
|
50
|
+
type="heading",
|
|
51
|
+
level=1,
|
|
52
|
+
text=f"Slide {slide_num}",
|
|
53
|
+
page=slide_num,
|
|
54
|
+
markdown_repr=f"# Slide {slide_num}",
|
|
55
|
+
confidence=1.0,
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
# 2. Iterate through shapes on the slide
|
|
59
|
+
for shape in slide.shapes:
|
|
60
|
+
# Skip title shape since already extracted
|
|
61
|
+
if shape == slide.shapes.title:
|
|
62
|
+
continue
|
|
63
|
+
# A. Handle Tables in slides
|
|
64
|
+
if shape.has_table:
|
|
65
|
+
table = shape.table
|
|
66
|
+
rows = []
|
|
67
|
+
for row in table.rows:
|
|
68
|
+
row_cells = [cell.text.strip() for cell in row.cells]
|
|
69
|
+
if any(row_cells):
|
|
70
|
+
rows.append(row_cells)
|
|
71
|
+
|
|
72
|
+
if not rows:
|
|
73
|
+
continue
|
|
74
|
+
|
|
75
|
+
headers = rows[0]
|
|
76
|
+
data_rows = rows[1:]
|
|
77
|
+
|
|
78
|
+
md_header = "| " + " | ".join(headers) + " |"
|
|
79
|
+
md_separator = "| " + " | ".join(["---"] * len(headers)) + " |"
|
|
80
|
+
md_rows = ["| " + " | ".join(r) + " |" for r in data_rows]
|
|
81
|
+
markdown_repr = "\n".join([md_header, md_separator] + md_rows)
|
|
82
|
+
|
|
83
|
+
yield Element(
|
|
84
|
+
type="table",
|
|
85
|
+
page=slide_num,
|
|
86
|
+
text=f"Slide {slide_num} Table",
|
|
87
|
+
data=TableData(headers=headers, rows=data_rows),
|
|
88
|
+
markdown_repr=markdown_repr,
|
|
89
|
+
confidence=1.0,
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
# B. Handle Text Frames (paragraphs and bullet points)
|
|
93
|
+
elif shape.has_text_frame:
|
|
94
|
+
for paragraph in shape.text_frame.paragraphs:
|
|
95
|
+
text = paragraph.text.strip()
|
|
96
|
+
if not text:
|
|
97
|
+
continue
|
|
98
|
+
|
|
99
|
+
is_bullet = paragraph.level > 0
|
|
100
|
+
elem_type = "list_item" if is_bullet else "paragraph"
|
|
101
|
+
prefix = " " * paragraph.level + "- " if is_bullet else ""
|
|
102
|
+
|
|
103
|
+
yield Element(
|
|
104
|
+
type=elem_type,
|
|
105
|
+
page=slide_num,
|
|
106
|
+
text=text,
|
|
107
|
+
markdown_repr=f"{prefix} {text}",
|
|
108
|
+
confidence=1.0,
|
|
109
|
+
)
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Iterator
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import ClassVar
|
|
6
|
+
|
|
7
|
+
from openpyxl import load_workbook
|
|
8
|
+
|
|
9
|
+
from universal_parser.core.router import register
|
|
10
|
+
from universal_parser.core.schema import Element, TableData
|
|
11
|
+
from universal_parser.core.sniffer import FileType
|
|
12
|
+
from universal_parser.extractors.base import BaseExtractor
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@register
|
|
16
|
+
class XlsxExtractor(BaseExtractor):
|
|
17
|
+
"""
|
|
18
|
+
Extractor for .xlsx files using openpyxl.
|
|
19
|
+
Handles:
|
|
20
|
+
- Memory-safe streaming using read_only=True
|
|
21
|
+
- Multiple sheets (each sheet is extracted as a Table element)
|
|
22
|
+
- Converts formula cells to values automatically using data_only=True
|
|
23
|
+
- Forward-fills empty/merged header names
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
supported_types: ClassVar[list[FileType]] = [FileType.XLSX]
|
|
27
|
+
|
|
28
|
+
def stream(self, path: str | Path) -> Iterator[Element]:
|
|
29
|
+
"""Stream sheets from an Excel workbook as Table elements."""
|
|
30
|
+
path_str = str(path)
|
|
31
|
+
|
|
32
|
+
try:
|
|
33
|
+
# read_only=True streams cells instead of loading the whole document.
|
|
34
|
+
# data_only=True evaluates Excel formulas and returns values.
|
|
35
|
+
wb = load_workbook(path_str, read_only=True, data_only=True)
|
|
36
|
+
|
|
37
|
+
except Exception: # noqa: BLE001
|
|
38
|
+
# Corrupted or password-protected file
|
|
39
|
+
return
|
|
40
|
+
|
|
41
|
+
try:
|
|
42
|
+
for sheet_name in wb.sheetnames:
|
|
43
|
+
sheet = wb[sheet_name]
|
|
44
|
+
|
|
45
|
+
# Extract rows from sheet
|
|
46
|
+
rows = []
|
|
47
|
+
for row in sheet.iter_rows(values_only=True):
|
|
48
|
+
# Clean the row data (convert None values to empty strings)
|
|
49
|
+
clean_row = [str(cell).strip() if cell is not None else "" for cell in row]
|
|
50
|
+
|
|
51
|
+
# Skip completely empty rows
|
|
52
|
+
if any(clean_row):
|
|
53
|
+
rows.append(clean_row)
|
|
54
|
+
|
|
55
|
+
if not rows:
|
|
56
|
+
continue
|
|
57
|
+
|
|
58
|
+
# Clean headers: handle merged cells by filling empty header slots
|
|
59
|
+
raw_headers = rows[0]
|
|
60
|
+
headers = []
|
|
61
|
+
last_seen_header = ""
|
|
62
|
+
for idx, h in enumerate(raw_headers):
|
|
63
|
+
if h:
|
|
64
|
+
headers.append(h)
|
|
65
|
+
last_seen_header = h
|
|
66
|
+
elif last_seen_header:
|
|
67
|
+
# Merged column continuation: e.g. "Q1 Revenue_2"
|
|
68
|
+
headers.append(f"{last_seen_header}_col{idx + 1}")
|
|
69
|
+
else:
|
|
70
|
+
headers.append(f"Column_{idx + 1}")
|
|
71
|
+
|
|
72
|
+
data_rows = rows[1:]
|
|
73
|
+
|
|
74
|
+
# Build a markdown representation of the spreadsheet
|
|
75
|
+
md_header = "| " + " | ".join(headers) + " |"
|
|
76
|
+
md_separator = "| " + " | ".join(["---"] * len(headers)) + " |"
|
|
77
|
+
md_rows = ["| " + " | ".join(r) + " |" for r in data_rows]
|
|
78
|
+
markdown_repr = f"### Sheet: {sheet_name}\n\n" + "\n".join(
|
|
79
|
+
[md_header, md_separator] + md_rows
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
yield Element(
|
|
83
|
+
type="table",
|
|
84
|
+
text=f"Sheet: {sheet_name}",
|
|
85
|
+
data=TableData(headers=headers, rows=data_rows),
|
|
86
|
+
markdown_repr=markdown_repr,
|
|
87
|
+
confidence=1.0,
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
except Exception: # noqa: BLE001
|
|
91
|
+
return
|
|
92
|
+
finally:
|
|
93
|
+
wb.close()
|
|
File without changes
|