markitdown-pro 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- markitdown_pro/__init__.py +1 -0
- markitdown_pro/common/__init__.py +0 -0
- markitdown_pro/common/logger.py +6 -0
- markitdown_pro/common/utils.py +59 -0
- markitdown_pro/conversion_pipeline.py +212 -0
- markitdown_pro/converters/__init__.py +0 -0
- markitdown_pro/converters/azure_docint.py +13 -0
- markitdown_pro/converters/base.py +26 -0
- markitdown_pro/converters/gpt4o_mini_vision.py +21 -0
- markitdown_pro/converters/markitdown_wrapper.py +44 -0
- markitdown_pro/converters/pymupdf_wrapper.py +42 -0
- markitdown_pro/converters/unstructured_wrapper.py +62 -0
- markitdown_pro/converters/youtube_wrapper.py +67 -0
- markitdown_pro/handlers/__init__.py +0 -0
- markitdown_pro/handlers/audio_handler.py +40 -0
- markitdown_pro/handlers/base_handler.py +16 -0
- markitdown_pro/handlers/email_handler.py +169 -0
- markitdown_pro/handlers/epub_handler.py +33 -0
- markitdown_pro/handlers/image_handler.py +48 -0
- markitdown_pro/handlers/ipynb_handler.py +31 -0
- markitdown_pro/handlers/markup_handler.py +75 -0
- markitdown_pro/handlers/office_handler.py +47 -0
- markitdown_pro/handlers/pdf_handler.py +122 -0
- markitdown_pro/handlers/pst_handler.py +153 -0
- markitdown_pro/handlers/tabular_handler.py +34 -0
- markitdown_pro/handlers/text_handler.py +38 -0
- markitdown_pro/services/__init__.py +0 -0
- markitdown_pro/services/azure_service.py +160 -0
- markitdown_pro/services/openai_services.py +209 -0
- markitdown_pro-0.1.0.dist-info/METADATA +367 -0
- markitdown_pro-0.1.0.dist-info/RECORD +33 -0
- markitdown_pro-0.1.0.dist-info/WHEEL +5 -0
- markitdown_pro-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import tempfile
|
|
3
|
+
from email.message import Message
|
|
4
|
+
from email.parser import BytesParser
|
|
5
|
+
from email.policy import default
|
|
6
|
+
from typing import Dict, List, Tuple
|
|
7
|
+
|
|
8
|
+
from ..common.logger import logger
|
|
9
|
+
from ..common.utils import clean_markdown
|
|
10
|
+
from .base_handler import BaseHandler
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class EmailHandler(BaseHandler):
|
|
14
|
+
extensions = frozenset([".eml", ".p7s"])
|
|
15
|
+
|
|
16
|
+
async def handle(self, file_path, *args, **kwargs) -> str:
|
|
17
|
+
logger.info(f"Processing email file: {file_path}")
|
|
18
|
+
|
|
19
|
+
try:
|
|
20
|
+
email_data = self._parse_email(file_path)
|
|
21
|
+
if not email_data:
|
|
22
|
+
return "# Error: Could not parse email file."
|
|
23
|
+
|
|
24
|
+
markdown_content = self._build_markdown(email_data)
|
|
25
|
+
return markdown_content
|
|
26
|
+
|
|
27
|
+
except Exception as e:
|
|
28
|
+
logger.error(f"Error processing email file: {file_path}: {e}")
|
|
29
|
+
return f"# Error processing email: {e}"
|
|
30
|
+
|
|
31
|
+
def _parse_email(self, file_path: str) -> Dict[str, any]:
|
|
32
|
+
"""
|
|
33
|
+
Parses the EML/P7S file and extracts relevant information.
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
file_path: The path to the email file.
|
|
37
|
+
|
|
38
|
+
Returns:
|
|
39
|
+
A dictionary containing extracted data:
|
|
40
|
+
{
|
|
41
|
+
"subject": str,
|
|
42
|
+
"from": str,
|
|
43
|
+
"to": str,
|
|
44
|
+
"date": str,
|
|
45
|
+
"body": str, # Plain text or HTML body (plain text preferred)
|
|
46
|
+
"attachments": List[Tuple[str, str]], # (filename, temp_file_path)
|
|
47
|
+
}
|
|
48
|
+
Returns an empty dictionary if parsing fails.
|
|
49
|
+
"""
|
|
50
|
+
try:
|
|
51
|
+
with open(file_path, "rb") as f:
|
|
52
|
+
# Use BytesParser with the 'default' policy for best compatibility
|
|
53
|
+
msg: Message = BytesParser(policy=default).parse(f)
|
|
54
|
+
|
|
55
|
+
subject = msg.get("Subject", "(No Subject)")
|
|
56
|
+
from_ = msg.get("From", "(Unknown Sender)")
|
|
57
|
+
to_ = msg.get("To", "(Unknown Recipient)")
|
|
58
|
+
date_ = msg.get("Date", "(Unknown Date)")
|
|
59
|
+
|
|
60
|
+
body = ""
|
|
61
|
+
attachments: List[Tuple[str, str]] = []
|
|
62
|
+
|
|
63
|
+
# Prefer plain text body
|
|
64
|
+
if msg.is_multipart():
|
|
65
|
+
for part in msg.walk():
|
|
66
|
+
content_type = part.get_content_type()
|
|
67
|
+
if part.get_content_disposition() == "attachment":
|
|
68
|
+
filename = part.get_filename()
|
|
69
|
+
if filename:
|
|
70
|
+
att_data = part.get_payload(decode=True)
|
|
71
|
+
# Create a temporary file for the attachment
|
|
72
|
+
with tempfile.NamedTemporaryFile(
|
|
73
|
+
delete=False, suffix=os.path.splitext(filename)[1]
|
|
74
|
+
) as tmp_file:
|
|
75
|
+
tmp_file.write(att_data)
|
|
76
|
+
attachments.append((filename, tmp_file.name))
|
|
77
|
+
|
|
78
|
+
elif content_type == "text/plain" and not body:
|
|
79
|
+
# Get the charset, default to utf-8 if not specified
|
|
80
|
+
charset = part.get_content_charset() or "utf-8"
|
|
81
|
+
try:
|
|
82
|
+
body = part.get_payload(decode=True).decode(charset, errors="replace")
|
|
83
|
+
except Exception as decode_err:
|
|
84
|
+
logger.warning(
|
|
85
|
+
f"Error decoding text/plain part: {decode_err}, using fallback",
|
|
86
|
+
)
|
|
87
|
+
body = part.get_payload(decode=True).decode("utf-8", errors="replace")
|
|
88
|
+
|
|
89
|
+
# If no plain text, try for HTML
|
|
90
|
+
if not body:
|
|
91
|
+
for part in msg.walk():
|
|
92
|
+
if part.get_content_type() == "text/html":
|
|
93
|
+
charset = part.get_content_charset() or "utf-8"
|
|
94
|
+
try:
|
|
95
|
+
body = part.get_payload(decode=True).decode(
|
|
96
|
+
charset, errors="replace"
|
|
97
|
+
)
|
|
98
|
+
break # Stop after finding the first HTML part
|
|
99
|
+
except Exception as decode_err:
|
|
100
|
+
logger.warning(
|
|
101
|
+
f"Error decoding text/html part: {decode_err}, using fallback",
|
|
102
|
+
)
|
|
103
|
+
body = part.get_payload(decode=True).decode(
|
|
104
|
+
"utf-8", errors="replace"
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
else: # Not multipart
|
|
108
|
+
content_type = msg.get_content_type()
|
|
109
|
+
if content_type == "text/plain":
|
|
110
|
+
charset = msg.get_content_charset() or "utf-8"
|
|
111
|
+
body = msg.get_payload(decode=True).decode(charset, "replace")
|
|
112
|
+
|
|
113
|
+
elif content_type == "text/html":
|
|
114
|
+
charset = msg.get_content_charset() or "utf-8"
|
|
115
|
+
body = msg.get_payload(decode=True).decode(charset, errors="replace")
|
|
116
|
+
|
|
117
|
+
return {
|
|
118
|
+
"subject": subject,
|
|
119
|
+
"from": from_,
|
|
120
|
+
"to": to_,
|
|
121
|
+
"date": date_,
|
|
122
|
+
"body": body,
|
|
123
|
+
"attachments": attachments,
|
|
124
|
+
}
|
|
125
|
+
except Exception as e:
|
|
126
|
+
logger.error(f"Error parsing email {file_path}: {e}")
|
|
127
|
+
return {}
|
|
128
|
+
|
|
129
|
+
def _build_markdown(self, email_data: Dict[str, any]) -> str:
|
|
130
|
+
"""
|
|
131
|
+
Builds the Markdown output from the extracted email data.
|
|
132
|
+
|
|
133
|
+
Args:
|
|
134
|
+
email_data: The dictionary returned by _parse_email.
|
|
135
|
+
|
|
136
|
+
Returns:
|
|
137
|
+
The complete Markdown string.
|
|
138
|
+
"""
|
|
139
|
+
# Use an f-string for more concise header formatting
|
|
140
|
+
markdown_parts = [
|
|
141
|
+
f"# Email: {email_data['subject']}",
|
|
142
|
+
f"**From:** {email_data['from']}",
|
|
143
|
+
f"**To:** {email_data['to']}",
|
|
144
|
+
f"**Date:** {email_data['date']}",
|
|
145
|
+
"", # Add an empty line for separation
|
|
146
|
+
]
|
|
147
|
+
|
|
148
|
+
body_content = email_data["body"]
|
|
149
|
+
if body_content:
|
|
150
|
+
markdown_parts.append("```")
|
|
151
|
+
markdown_parts.append(body_content) # Add the email body
|
|
152
|
+
markdown_parts.append("```")
|
|
153
|
+
|
|
154
|
+
for filename, filepath in email_data["attachments"]:
|
|
155
|
+
markdown_parts.append(f"\n## Attachment: {filename}\n")
|
|
156
|
+
try:
|
|
157
|
+
# todo
|
|
158
|
+
attachment_markdown = None
|
|
159
|
+
markdown_parts.append(attachment_markdown)
|
|
160
|
+
except Exception as e:
|
|
161
|
+
logger.error(f"Error converting attachment '{filename}' in email: {e}")
|
|
162
|
+
markdown_parts.append(f"[Error converting attachment: {e}]")
|
|
163
|
+
finally:
|
|
164
|
+
try:
|
|
165
|
+
os.remove(filepath) # Clean up the temporary attachment file
|
|
166
|
+
except OSError as e:
|
|
167
|
+
logger.warning(f"Could not remove temporary attachment file '{filepath}': {e}")
|
|
168
|
+
|
|
169
|
+
return clean_markdown("\n".join(markdown_parts))
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
from ..common.logger import logger
|
|
2
|
+
from ..common.utils import ensure_minimum_content
|
|
3
|
+
from ..converters.unstructured_wrapper import UnstructuredWrapper
|
|
4
|
+
from .base_handler import BaseHandler
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class EPUBHandler(BaseHandler):
|
|
8
|
+
extensions = frozenset([".epub"])
|
|
9
|
+
|
|
10
|
+
def __init__(self, *args, **kwargs):
|
|
11
|
+
super().__init__(*args, **kwargs)
|
|
12
|
+
self.unstructured = UnstructuredWrapper()
|
|
13
|
+
|
|
14
|
+
async def handle(self, file_path, *args, **kwargs) -> str:
|
|
15
|
+
"""
|
|
16
|
+
Handles EPUB files by converting them to Markdown using Unstructured.
|
|
17
|
+
|
|
18
|
+
Args:
|
|
19
|
+
file_path: Path to the .epub file.
|
|
20
|
+
|
|
21
|
+
Returns:
|
|
22
|
+
Markdown string representing the EPUB content, or an error message.
|
|
23
|
+
"""
|
|
24
|
+
logger.info(f"Processing EPUB file {file_path}")
|
|
25
|
+
try:
|
|
26
|
+
md_content = await self.unstructured.convert(file_path) # added await
|
|
27
|
+
if md_content and ensure_minimum_content(md_content):
|
|
28
|
+
return md_content
|
|
29
|
+
else:
|
|
30
|
+
raise RuntimeError(f"EPUB conversion failed or insufficient content: {file_path}")
|
|
31
|
+
except Exception as e:
|
|
32
|
+
logger.error(f"Error handling EPUB file '{file_path}': {e}")
|
|
33
|
+
return "# Error al procesar EPUB"
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
from ..common.logger import logger
|
|
2
|
+
from ..common.utils import ensure_minimum_content
|
|
3
|
+
from ..converters.gpt4o_mini_vision import GPT4oMiniVisionWrapper
|
|
4
|
+
from .base_handler import BaseHandler
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class ImageHandler(BaseHandler):
|
|
8
|
+
extensions = frozenset(
|
|
9
|
+
[
|
|
10
|
+
".bmp",
|
|
11
|
+
".gif",
|
|
12
|
+
".heic",
|
|
13
|
+
".jpeg",
|
|
14
|
+
".jpg",
|
|
15
|
+
".png",
|
|
16
|
+
".prn",
|
|
17
|
+
".svg",
|
|
18
|
+
".tiff",
|
|
19
|
+
".webp",
|
|
20
|
+
".heif",
|
|
21
|
+
]
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
def __init__(self, *args, **kwargs):
|
|
25
|
+
super().__init__(*args, **kwargs)
|
|
26
|
+
self.gpt4o_mini_vision = GPT4oMiniVisionWrapper()
|
|
27
|
+
|
|
28
|
+
async def handle(self, file_path, *args, **kwargs) -> str:
|
|
29
|
+
"""
|
|
30
|
+
Handles image files by converting them to Markdown using GPT-4o-mini Vision.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
file_path: Path to the image file.
|
|
34
|
+
|
|
35
|
+
Returns:
|
|
36
|
+
Markdown string representing the image content (OCR and analysis),
|
|
37
|
+
or an error message.
|
|
38
|
+
"""
|
|
39
|
+
logger.info(f"Processing image file: {file_path}")
|
|
40
|
+
try:
|
|
41
|
+
md_content = await self.gpt4o_mini_vision.convert(file_path)
|
|
42
|
+
if md_content and ensure_minimum_content(md_content):
|
|
43
|
+
return md_content
|
|
44
|
+
else:
|
|
45
|
+
raise RuntimeError(f"Image conversion failed or insufficient content: {file_path}")
|
|
46
|
+
except Exception as e:
|
|
47
|
+
logger.error(f"Error handling image file '{file_path}': {e}")
|
|
48
|
+
return "# Error al procesar imagen"
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
import nbformat
|
|
2
|
+
|
|
3
|
+
from ..common import utils
|
|
4
|
+
from ..common.logger import logger
|
|
5
|
+
from .base_handler import BaseHandler
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class IpynbHandler(BaseHandler):
|
|
9
|
+
"""Handler for Jupyter notebooks (.ipynb)."""
|
|
10
|
+
|
|
11
|
+
extensions = frozenset([".ipynb"])
|
|
12
|
+
|
|
13
|
+
async def handle(self, file_path: str, *args, **kwargs) -> str | None:
|
|
14
|
+
logger.info(f"Processing notebook: {file_path}")
|
|
15
|
+
try:
|
|
16
|
+
nb = nbformat.read(file_path, as_version=4)
|
|
17
|
+
cells_content = []
|
|
18
|
+
for cell in nb.cells:
|
|
19
|
+
if cell.cell_type == "markdown":
|
|
20
|
+
cells_content.append(cell.source)
|
|
21
|
+
elif cell.cell_type == "code":
|
|
22
|
+
cells_content.append("```python\n" + cell.source + "\n```")
|
|
23
|
+
result = "\n\n".join(cells_content)
|
|
24
|
+
|
|
25
|
+
if utils.ensure_minimum_content(result):
|
|
26
|
+
return result
|
|
27
|
+
else:
|
|
28
|
+
return None
|
|
29
|
+
except Exception as e:
|
|
30
|
+
logger.error(f"Error processing notebook {file_path}: {e}")
|
|
31
|
+
return None
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import os
|
|
3
|
+
|
|
4
|
+
import yaml
|
|
5
|
+
from bs4 import BeautifulSoup
|
|
6
|
+
|
|
7
|
+
try:
|
|
8
|
+
import chardet
|
|
9
|
+
except ImportError:
|
|
10
|
+
chardet = None
|
|
11
|
+
|
|
12
|
+
from ..common.logger import logger
|
|
13
|
+
from ..common.utils import ensure_minimum_content
|
|
14
|
+
from .base_handler import BaseHandler
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class MarkupHandler(BaseHandler):
|
|
18
|
+
"""Handler for .html, .xml, .json, .ndjson, .yaml, .yml files."""
|
|
19
|
+
|
|
20
|
+
extensions = frozenset([".html", ".htm", ".xml", ".json", ".ndjson", ".yaml", ".yml"])
|
|
21
|
+
|
|
22
|
+
async def handle(self, file_path: str, *args, **kwargs) -> str:
|
|
23
|
+
logger.info(f"Processing markup file: {file_path}")
|
|
24
|
+
try:
|
|
25
|
+
ext = os.path.splitext(file_path)[1].lower()
|
|
26
|
+
|
|
27
|
+
# Detect encoding
|
|
28
|
+
encoding = "utf-8" # Default
|
|
29
|
+
if chardet:
|
|
30
|
+
with open(file_path, "rb") as f:
|
|
31
|
+
raw_data = f.read()
|
|
32
|
+
result = chardet.detect(raw_data)
|
|
33
|
+
encoding = result["encoding"]
|
|
34
|
+
logger.debug(f"Detected encoding: {encoding}")
|
|
35
|
+
else:
|
|
36
|
+
logger.warning("chardet not available, assuming UTF-8 encoding.")
|
|
37
|
+
|
|
38
|
+
with open(file_path, "r", encoding=encoding) as f:
|
|
39
|
+
content = f.read()
|
|
40
|
+
|
|
41
|
+
if ext in [".html", ".htm"]:
|
|
42
|
+
soup = BeautifulSoup(content, "html.parser")
|
|
43
|
+
text = soup.get_text(separator="\n")
|
|
44
|
+
elif ext == ".xml":
|
|
45
|
+
soup = BeautifulSoup(content, "xml")
|
|
46
|
+
text = soup.get_text(separator="\n")
|
|
47
|
+
elif ext in [".json", ".ndjson"]:
|
|
48
|
+
try:
|
|
49
|
+
# Attempt to parse as complete JSON
|
|
50
|
+
data = json.loads(content)
|
|
51
|
+
text = json.dumps(data, indent=2, ensure_ascii=False)
|
|
52
|
+
except Exception:
|
|
53
|
+
# If not, process line by line (ndjson case)
|
|
54
|
+
lines = content.splitlines()
|
|
55
|
+
parsed_lines = []
|
|
56
|
+
for line in lines:
|
|
57
|
+
try:
|
|
58
|
+
obj = json.loads(line)
|
|
59
|
+
parsed_lines.append(json.dumps(obj, indent=2, ensure_ascii=False))
|
|
60
|
+
except Exception:
|
|
61
|
+
parsed_lines.append(line)
|
|
62
|
+
text = "\n".join(parsed_lines)
|
|
63
|
+
elif ext in [".yaml", ".yml"]:
|
|
64
|
+
data = yaml.safe_load(content)
|
|
65
|
+
text = yaml.dump(data, allow_unicode=True)
|
|
66
|
+
else:
|
|
67
|
+
text = content
|
|
68
|
+
|
|
69
|
+
if ensure_minimum_content(text):
|
|
70
|
+
return text
|
|
71
|
+
else:
|
|
72
|
+
raise RuntimeError(f"Insufficient content after conversion: {file_path}")
|
|
73
|
+
except Exception as e:
|
|
74
|
+
logger.error(f"Error processing markup file {file_path}: {e}")
|
|
75
|
+
raise
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
from ..common.logger import logger
|
|
2
|
+
from ..common.utils import ensure_minimum_content
|
|
3
|
+
from ..converters.azure_docint import AzureDocIntWrapper
|
|
4
|
+
from ..converters.unstructured_wrapper import UnstructuredWrapper
|
|
5
|
+
from .base_handler import BaseHandler
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class OfficeHandler(BaseHandler):
|
|
9
|
+
extensions = frozenset([".doc", ".docx", ".odt", ".rtf", ".ppt", ".pptx"])
|
|
10
|
+
|
|
11
|
+
def __init__(self, *args, **kwargs):
|
|
12
|
+
super().__init__(*args, **kwargs)
|
|
13
|
+
self.azure_docint = AzureDocIntWrapper()
|
|
14
|
+
self.unstructured = UnstructuredWrapper()
|
|
15
|
+
|
|
16
|
+
async def handle(self, file_path, *args, **kwargs) -> str:
|
|
17
|
+
"""
|
|
18
|
+
Handles Office documents (.doc, .docx, .odt, .rtf, .ppt, .pptx) by
|
|
19
|
+
first trying Azure Document Intelligence and falling back to Unstructured.
|
|
20
|
+
|
|
21
|
+
Args:
|
|
22
|
+
file_path: Path to the Office document file.
|
|
23
|
+
|
|
24
|
+
Returns:
|
|
25
|
+
Markdown string representing the document content, or an error message.
|
|
26
|
+
"""
|
|
27
|
+
logger.info(f"Processing Office document: {file_path}")
|
|
28
|
+
try:
|
|
29
|
+
# First try with Azure Document Intelligence
|
|
30
|
+
logger.info(f"Attempting conversion with Azure Document Intelligence for: {file_path}")
|
|
31
|
+
md_content = await self.azure_docint.convert(file_path)
|
|
32
|
+
if md_content and ensure_minimum_content(md_content):
|
|
33
|
+
return md_content
|
|
34
|
+
|
|
35
|
+
# Fallback to Unstructured if Azure Doc Intelligence fails or returns insufficient content
|
|
36
|
+
logger.info(f"Falling back to Unstructured for: {file_path}")
|
|
37
|
+
md_content = await self.unstructured.convert(file_path)
|
|
38
|
+
if md_content and ensure_minimum_content(md_content):
|
|
39
|
+
return md_content
|
|
40
|
+
|
|
41
|
+
raise RuntimeError(
|
|
42
|
+
f"Office document conversion failed with both Azure Doc Intelligence and Unstructured: {file_path}"
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
except Exception as e:
|
|
46
|
+
logger.error(f"Error handling Office document '{file_path}': {e}")
|
|
47
|
+
return "# Error al procesar documento de Office"
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
from enum import Enum
|
|
2
|
+
|
|
3
|
+
import fitz
|
|
4
|
+
|
|
5
|
+
from ..common.logger import logger
|
|
6
|
+
from ..common.utils import ensure_minimum_content
|
|
7
|
+
from ..converters.azure_docint import AzureDocIntWrapper
|
|
8
|
+
from ..converters.gpt4o_mini_vision import GPT4oMiniVisionWrapper
|
|
9
|
+
from ..converters.markitdown_wrapper import MarkitDownWrapper
|
|
10
|
+
from ..converters.pymupdf_wrapper import PyMuPDFWrapper
|
|
11
|
+
from ..converters.unstructured_wrapper import UnstructuredWrapper
|
|
12
|
+
from .base_handler import BaseHandler
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class PDFType(Enum):
|
|
16
|
+
"""
|
|
17
|
+
Tipos de PDF detectados por el PDFHandler.
|
|
18
|
+
|
|
19
|
+
- TEXT_ONLY: PDF que contiene solo texto, sin imágenes.
|
|
20
|
+
- TEXT_PLUS_IMAGES: PDF que contiene texto e imágenes.
|
|
21
|
+
- ALL_IMAGES: PDF que contiene solo imágenes (PDF escaneado).
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
TEXT_ONLY = "TEXT_ONLY"
|
|
25
|
+
TEXT_PLUS_IMAGES = "TEXT_PLUS_IMAGES"
|
|
26
|
+
ALL_IMAGES = "ALL_IMAGES"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class PDFHandler(BaseHandler):
|
|
30
|
+
extensions = frozenset([".pdf"])
|
|
31
|
+
|
|
32
|
+
def __init__(self, *args, **kwargs):
|
|
33
|
+
super().__init__(*args, **kwargs)
|
|
34
|
+
self.markitdown = MarkitDownWrapper()
|
|
35
|
+
self.unstructured = UnstructuredWrapper()
|
|
36
|
+
self.azure_docint = AzureDocIntWrapper()
|
|
37
|
+
self.gpt4o_mini_vision = GPT4oMiniVisionWrapper()
|
|
38
|
+
self.pymu = PyMuPDFWrapper()
|
|
39
|
+
|
|
40
|
+
self.text_pipeline = [
|
|
41
|
+
self.markitdown,
|
|
42
|
+
self.unstructured,
|
|
43
|
+
self.pymu,
|
|
44
|
+
self.azure_docint,
|
|
45
|
+
]
|
|
46
|
+
self.image_pipeline = [
|
|
47
|
+
self.gpt4o_mini_vision,
|
|
48
|
+
]
|
|
49
|
+
|
|
50
|
+
async def handle(self, file_path, *args, **kwargs):
|
|
51
|
+
try:
|
|
52
|
+
pdf_type = await self._detect_pdf_type(file_path)
|
|
53
|
+
|
|
54
|
+
if pdf_type == PDFType.TEXT_ONLY:
|
|
55
|
+
pipeline = self.text_pipeline
|
|
56
|
+
elif pdf_type == PDFType.ALL_IMAGES:
|
|
57
|
+
pipeline = self.image_pipeline
|
|
58
|
+
elif pdf_type == PDFType.TEXT_PLUS_IMAGES:
|
|
59
|
+
pipeline = self.text_pipeline + self.image_pipeline
|
|
60
|
+
else:
|
|
61
|
+
pipeline = self.text_pipeline
|
|
62
|
+
|
|
63
|
+
for converter in pipeline:
|
|
64
|
+
logger.info(f"Trying {converter.name} for PDF {file_path}")
|
|
65
|
+
try:
|
|
66
|
+
md_content = await converter.convert(file_path)
|
|
67
|
+
if md_content and ensure_minimum_content(md_content):
|
|
68
|
+
return md_content
|
|
69
|
+
except Exception as e:
|
|
70
|
+
logger.error(f"Converter {converter.name} failed for PDF {file_path}: {e}")
|
|
71
|
+
|
|
72
|
+
raise RuntimeError(f"PDF conversion failed with all converters for {file_path}")
|
|
73
|
+
|
|
74
|
+
except Exception as e:
|
|
75
|
+
logger.error(f"Error handling PDF '{file_path}': {e}")
|
|
76
|
+
return None
|
|
77
|
+
|
|
78
|
+
async def _detect_pdf_type(self, file_path: str) -> PDFType:
|
|
79
|
+
"""
|
|
80
|
+
Detect the type of PDF file based on its content.
|
|
81
|
+
"""
|
|
82
|
+
min_text_length_threshold = 50
|
|
83
|
+
|
|
84
|
+
try:
|
|
85
|
+
|
|
86
|
+
async def open_and_process_doc():
|
|
87
|
+
with fitz.open(file_path) as doc:
|
|
88
|
+
total_pages = doc.page_count
|
|
89
|
+
pages_with_text = 0
|
|
90
|
+
pages_with_images = 0
|
|
91
|
+
|
|
92
|
+
for page_index in range(total_pages):
|
|
93
|
+
page = doc.load_page(page_index)
|
|
94
|
+
page_text = page.get_text().strip()
|
|
95
|
+
if len(page_text) >= min_text_length_threshold:
|
|
96
|
+
pages_with_text += 1
|
|
97
|
+
images = page.get_images(full=True)
|
|
98
|
+
if images:
|
|
99
|
+
pages_with_images += 1
|
|
100
|
+
|
|
101
|
+
logger.debug(f"Pages with text: {pages_with_text}/{total_pages}")
|
|
102
|
+
logger.debug(f"Pages with images: {pages_with_images}/{total_pages}")
|
|
103
|
+
|
|
104
|
+
is_text_only = pages_with_text == total_pages and pages_with_images == 0
|
|
105
|
+
is_all_images = pages_with_images == total_pages and pages_with_text == 0
|
|
106
|
+
has_text_and_images = pages_with_text > 0 and pages_with_images > 0
|
|
107
|
+
|
|
108
|
+
if is_text_only:
|
|
109
|
+
return PDFType.TEXT_ONLY
|
|
110
|
+
elif is_all_images:
|
|
111
|
+
return PDFType.ALL_IMAGES
|
|
112
|
+
elif has_text_and_images:
|
|
113
|
+
return PDFType.TEXT_PLUS_IMAGES
|
|
114
|
+
else:
|
|
115
|
+
return PDFType.TEXT_PLUS_IMAGES # Fallback
|
|
116
|
+
|
|
117
|
+
return await open_and_process_doc() # Run in thread
|
|
118
|
+
except Exception as e:
|
|
119
|
+
logger.error(f"Error analyzing PDF '{file_path}': {e}")
|
|
120
|
+
# Important to re-raise the exception after logging, so the
|
|
121
|
+
# caller knows something went wrong *during the analysis*.
|
|
122
|
+
raise
|
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
import os
|
|
2
|
+
from typing import Optional
|
|
3
|
+
|
|
4
|
+
from .base_handler import BaseHandler
|
|
5
|
+
|
|
6
|
+
try:
|
|
7
|
+
from libratom.lib.pff import PffArchive
|
|
8
|
+
|
|
9
|
+
HAS_LIBRATOM = True
|
|
10
|
+
except ImportError:
|
|
11
|
+
HAS_LIBRATOM = False
|
|
12
|
+
|
|
13
|
+
from ..common.logger import logger
|
|
14
|
+
from ..common.utils import clean_markdown, ensure_minimum_content
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class PSTHandler(BaseHandler):
|
|
18
|
+
extensions = frozenset([".pst"])
|
|
19
|
+
|
|
20
|
+
async def handle(self, file_path, *args, **kwargs) -> str:
|
|
21
|
+
"""
|
|
22
|
+
Parses a PST file using libratom, extracts messages and attachments,
|
|
23
|
+
and converts the content to Markdown. Recursively processes attachments.
|
|
24
|
+
|
|
25
|
+
Args:
|
|
26
|
+
file_path: Path to the .pst file.
|
|
27
|
+
|
|
28
|
+
Returns:
|
|
29
|
+
Markdown string representing the PST content, or an error message.
|
|
30
|
+
"""
|
|
31
|
+
if not HAS_LIBRATOM:
|
|
32
|
+
logger.error("libratom is not installed. PST processing is disabled.")
|
|
33
|
+
return "# Error: libratom not installed. Cannot process PST files."
|
|
34
|
+
|
|
35
|
+
logger.info(f"Processing PST file: {file_path}")
|
|
36
|
+
try:
|
|
37
|
+
markdown_content = self._process_pst(file_path)
|
|
38
|
+
if markdown_content:
|
|
39
|
+
return markdown_content
|
|
40
|
+
else:
|
|
41
|
+
return "# PST Archive\n\n(No messages found or insufficient content.)"
|
|
42
|
+
except Exception as e:
|
|
43
|
+
logger.error(f"Error processing PST file: {file_path}: {e}")
|
|
44
|
+
return f"# Error processing PST file: {e}"
|
|
45
|
+
|
|
46
|
+
def _process_pst(self, file_path: str) -> Optional[str]:
|
|
47
|
+
"""
|
|
48
|
+
Parses the PST file, extracts messages and attachments, and converts to Markdown.
|
|
49
|
+
|
|
50
|
+
Args:
|
|
51
|
+
file_path: Path to the PST file.
|
|
52
|
+
|
|
53
|
+
Returns:
|
|
54
|
+
Markdown string, or None if no messages are found or content is insufficient.
|
|
55
|
+
"""
|
|
56
|
+
if not os.path.isfile(file_path):
|
|
57
|
+
logger.error(f"PST file not found: {file_path}")
|
|
58
|
+
return None
|
|
59
|
+
|
|
60
|
+
all_md_parts = [f"# PST Archive: {os.path.basename(file_path)}\n"]
|
|
61
|
+
|
|
62
|
+
try:
|
|
63
|
+
with PffArchive(file_path) as archive:
|
|
64
|
+
for folder in archive.folders():
|
|
65
|
+
if not folder.name: # Skip folders with no name
|
|
66
|
+
continue
|
|
67
|
+
|
|
68
|
+
# Add folder information, handling None folder names
|
|
69
|
+
all_md_parts.append(f"\n## Folder: {folder.name or '(Unnamed Folder)'}\n")
|
|
70
|
+
message_count = 0
|
|
71
|
+
|
|
72
|
+
for message in folder.messages():
|
|
73
|
+
message_count += 1
|
|
74
|
+
try:
|
|
75
|
+
message_md = self._process_message(message, message_count)
|
|
76
|
+
if message_md:
|
|
77
|
+
all_md_parts.extend(message_md)
|
|
78
|
+
except Exception as e:
|
|
79
|
+
logger.error(
|
|
80
|
+
f"Error processing message {message_count} in folder {folder.name}: {e}"
|
|
81
|
+
)
|
|
82
|
+
all_md_parts.append(
|
|
83
|
+
f"### Error processing message {message_count}: {e}"
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
final_md = clean_markdown("\n\n".join(all_md_parts))
|
|
87
|
+
return final_md if ensure_minimum_content(final_md) else None
|
|
88
|
+
|
|
89
|
+
except Exception as e:
|
|
90
|
+
logger.error(f"Error opening or processing PST archive {file_path}: {e}")
|
|
91
|
+
return None
|
|
92
|
+
|
|
93
|
+
def _process_message(self, message, message_count: int) -> Optional[list[str]]:
|
|
94
|
+
"""
|
|
95
|
+
Processes a single message from the PST archive.
|
|
96
|
+
|
|
97
|
+
Args:
|
|
98
|
+
message: The message object from libratom.
|
|
99
|
+
message_count: The message number within the folder (for display).
|
|
100
|
+
|
|
101
|
+
Returns:
|
|
102
|
+
A list of Markdown strings representing the message, or None on error.
|
|
103
|
+
"""
|
|
104
|
+
try:
|
|
105
|
+
subject = message.subject or "(No Subject)"
|
|
106
|
+
sender = "Unknown Sender"
|
|
107
|
+
date_ = "Unknown Date"
|
|
108
|
+
|
|
109
|
+
# Extract headers safely, handling potential errors
|
|
110
|
+
try:
|
|
111
|
+
headers = message.transport_headers
|
|
112
|
+
if headers:
|
|
113
|
+
if isinstance(headers, bytes):
|
|
114
|
+
headers = headers.decode(errors="replace")
|
|
115
|
+
for line in headers.splitlines():
|
|
116
|
+
if line.lower().startswith("from:"):
|
|
117
|
+
sender = line.split(":", 1)[1].strip()
|
|
118
|
+
elif line.lower().startswith("date:"):
|
|
119
|
+
date_ = line.split(":", 1)[1].strip()
|
|
120
|
+
except Exception as header_err:
|
|
121
|
+
logger.warning(f"Error parsing headers: {header_err}")
|
|
122
|
+
|
|
123
|
+
# Extract the body, handling different encodings and body types
|
|
124
|
+
body_content = ""
|
|
125
|
+
try:
|
|
126
|
+
if message.plain_text_body:
|
|
127
|
+
body_content = message.plain_text_body.decode(errors="replace")
|
|
128
|
+
elif message.html_body:
|
|
129
|
+
body_content = message.html_body.decode(errors="replace")
|
|
130
|
+
elif message.rtf_body:
|
|
131
|
+
body_content = message.rtf_body.decode(errors="replace")
|
|
132
|
+
|
|
133
|
+
except Exception as body_err:
|
|
134
|
+
logger.warning(f"Error decoding message body: {body_err}")
|
|
135
|
+
|
|
136
|
+
message_md_parts = [
|
|
137
|
+
f"### Message {message_count}",
|
|
138
|
+
f"**Subject:** {subject}",
|
|
139
|
+
f"**From:** {sender}",
|
|
140
|
+
f"**Date:** {date_}",
|
|
141
|
+
"",
|
|
142
|
+
"```",
|
|
143
|
+
body_content.strip() if body_content else "[No body text]",
|
|
144
|
+
"```",
|
|
145
|
+
]
|
|
146
|
+
|
|
147
|
+
# Handle attachments todo: handle attachments
|
|
148
|
+
|
|
149
|
+
return message_md_parts
|
|
150
|
+
|
|
151
|
+
except Exception as e:
|
|
152
|
+
logger.error(f"Error processing individual message: {e}")
|
|
153
|
+
return None
|