markitdown-pro 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. markitdown_pro/__init__.py +1 -0
  2. markitdown_pro/common/__init__.py +0 -0
  3. markitdown_pro/common/logger.py +6 -0
  4. markitdown_pro/common/utils.py +59 -0
  5. markitdown_pro/conversion_pipeline.py +212 -0
  6. markitdown_pro/converters/__init__.py +0 -0
  7. markitdown_pro/converters/azure_docint.py +13 -0
  8. markitdown_pro/converters/base.py +26 -0
  9. markitdown_pro/converters/gpt4o_mini_vision.py +21 -0
  10. markitdown_pro/converters/markitdown_wrapper.py +44 -0
  11. markitdown_pro/converters/pymupdf_wrapper.py +42 -0
  12. markitdown_pro/converters/unstructured_wrapper.py +62 -0
  13. markitdown_pro/converters/youtube_wrapper.py +67 -0
  14. markitdown_pro/handlers/__init__.py +0 -0
  15. markitdown_pro/handlers/audio_handler.py +40 -0
  16. markitdown_pro/handlers/base_handler.py +16 -0
  17. markitdown_pro/handlers/email_handler.py +169 -0
  18. markitdown_pro/handlers/epub_handler.py +33 -0
  19. markitdown_pro/handlers/image_handler.py +48 -0
  20. markitdown_pro/handlers/ipynb_handler.py +31 -0
  21. markitdown_pro/handlers/markup_handler.py +75 -0
  22. markitdown_pro/handlers/office_handler.py +47 -0
  23. markitdown_pro/handlers/pdf_handler.py +122 -0
  24. markitdown_pro/handlers/pst_handler.py +153 -0
  25. markitdown_pro/handlers/tabular_handler.py +34 -0
  26. markitdown_pro/handlers/text_handler.py +38 -0
  27. markitdown_pro/services/__init__.py +0 -0
  28. markitdown_pro/services/azure_service.py +160 -0
  29. markitdown_pro/services/openai_services.py +209 -0
  30. markitdown_pro-0.1.0.dist-info/METADATA +367 -0
  31. markitdown_pro-0.1.0.dist-info/RECORD +33 -0
  32. markitdown_pro-0.1.0.dist-info/WHEEL +5 -0
  33. markitdown_pro-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,169 @@
1
+ import os
2
+ import tempfile
3
+ from email.message import Message
4
+ from email.parser import BytesParser
5
+ from email.policy import default
6
+ from typing import Dict, List, Tuple
7
+
8
+ from ..common.logger import logger
9
+ from ..common.utils import clean_markdown
10
+ from .base_handler import BaseHandler
11
+
12
+
13
+ class EmailHandler(BaseHandler):
14
+ extensions = frozenset([".eml", ".p7s"])
15
+
16
+ async def handle(self, file_path, *args, **kwargs) -> str:
17
+ logger.info(f"Processing email file: {file_path}")
18
+
19
+ try:
20
+ email_data = self._parse_email(file_path)
21
+ if not email_data:
22
+ return "# Error: Could not parse email file."
23
+
24
+ markdown_content = self._build_markdown(email_data)
25
+ return markdown_content
26
+
27
+ except Exception as e:
28
+ logger.error(f"Error processing email file: {file_path}: {e}")
29
+ return f"# Error processing email: {e}"
30
+
31
+ def _parse_email(self, file_path: str) -> Dict[str, any]:
32
+ """
33
+ Parses the EML/P7S file and extracts relevant information.
34
+
35
+ Args:
36
+ file_path: The path to the email file.
37
+
38
+ Returns:
39
+ A dictionary containing extracted data:
40
+ {
41
+ "subject": str,
42
+ "from": str,
43
+ "to": str,
44
+ "date": str,
45
+ "body": str, # Plain text or HTML body (plain text preferred)
46
+ "attachments": List[Tuple[str, str]], # (filename, temp_file_path)
47
+ }
48
+ Returns an empty dictionary if parsing fails.
49
+ """
50
+ try:
51
+ with open(file_path, "rb") as f:
52
+ # Use BytesParser with the 'default' policy for best compatibility
53
+ msg: Message = BytesParser(policy=default).parse(f)
54
+
55
+ subject = msg.get("Subject", "(No Subject)")
56
+ from_ = msg.get("From", "(Unknown Sender)")
57
+ to_ = msg.get("To", "(Unknown Recipient)")
58
+ date_ = msg.get("Date", "(Unknown Date)")
59
+
60
+ body = ""
61
+ attachments: List[Tuple[str, str]] = []
62
+
63
+ # Prefer plain text body
64
+ if msg.is_multipart():
65
+ for part in msg.walk():
66
+ content_type = part.get_content_type()
67
+ if part.get_content_disposition() == "attachment":
68
+ filename = part.get_filename()
69
+ if filename:
70
+ att_data = part.get_payload(decode=True)
71
+ # Create a temporary file for the attachment
72
+ with tempfile.NamedTemporaryFile(
73
+ delete=False, suffix=os.path.splitext(filename)[1]
74
+ ) as tmp_file:
75
+ tmp_file.write(att_data)
76
+ attachments.append((filename, tmp_file.name))
77
+
78
+ elif content_type == "text/plain" and not body:
79
+ # Get the charset, default to utf-8 if not specified
80
+ charset = part.get_content_charset() or "utf-8"
81
+ try:
82
+ body = part.get_payload(decode=True).decode(charset, errors="replace")
83
+ except Exception as decode_err:
84
+ logger.warning(
85
+ f"Error decoding text/plain part: {decode_err}, using fallback",
86
+ )
87
+ body = part.get_payload(decode=True).decode("utf-8", errors="replace")
88
+
89
+ # If no plain text, try for HTML
90
+ if not body:
91
+ for part in msg.walk():
92
+ if part.get_content_type() == "text/html":
93
+ charset = part.get_content_charset() or "utf-8"
94
+ try:
95
+ body = part.get_payload(decode=True).decode(
96
+ charset, errors="replace"
97
+ )
98
+ break # Stop after finding the first HTML part
99
+ except Exception as decode_err:
100
+ logger.warning(
101
+ f"Error decoding text/html part: {decode_err}, using fallback",
102
+ )
103
+ body = part.get_payload(decode=True).decode(
104
+ "utf-8", errors="replace"
105
+ )
106
+
107
+ else: # Not multipart
108
+ content_type = msg.get_content_type()
109
+ if content_type == "text/plain":
110
+ charset = msg.get_content_charset() or "utf-8"
111
+ body = msg.get_payload(decode=True).decode(charset, "replace")
112
+
113
+ elif content_type == "text/html":
114
+ charset = msg.get_content_charset() or "utf-8"
115
+ body = msg.get_payload(decode=True).decode(charset, errors="replace")
116
+
117
+ return {
118
+ "subject": subject,
119
+ "from": from_,
120
+ "to": to_,
121
+ "date": date_,
122
+ "body": body,
123
+ "attachments": attachments,
124
+ }
125
+ except Exception as e:
126
+ logger.error(f"Error parsing email {file_path}: {e}")
127
+ return {}
128
+
129
+ def _build_markdown(self, email_data: Dict[str, any]) -> str:
130
+ """
131
+ Builds the Markdown output from the extracted email data.
132
+
133
+ Args:
134
+ email_data: The dictionary returned by _parse_email.
135
+
136
+ Returns:
137
+ The complete Markdown string.
138
+ """
139
+ # Use an f-string for more concise header formatting
140
+ markdown_parts = [
141
+ f"# Email: {email_data['subject']}",
142
+ f"**From:** {email_data['from']}",
143
+ f"**To:** {email_data['to']}",
144
+ f"**Date:** {email_data['date']}",
145
+ "", # Add an empty line for separation
146
+ ]
147
+
148
+ body_content = email_data["body"]
149
+ if body_content:
150
+ markdown_parts.append("```")
151
+ markdown_parts.append(body_content) # Add the email body
152
+ markdown_parts.append("```")
153
+
154
+ for filename, filepath in email_data["attachments"]:
155
+ markdown_parts.append(f"\n## Attachment: {filename}\n")
156
+ try:
157
+ # todo
158
+ attachment_markdown = None
159
+ markdown_parts.append(attachment_markdown)
160
+ except Exception as e:
161
+ logger.error(f"Error converting attachment '{filename}' in email: {e}")
162
+ markdown_parts.append(f"[Error converting attachment: {e}]")
163
+ finally:
164
+ try:
165
+ os.remove(filepath) # Clean up the temporary attachment file
166
+ except OSError as e:
167
+ logger.warning(f"Could not remove temporary attachment file '{filepath}': {e}")
168
+
169
+ return clean_markdown("\n".join(markdown_parts))
@@ -0,0 +1,33 @@
1
+ from ..common.logger import logger
2
+ from ..common.utils import ensure_minimum_content
3
+ from ..converters.unstructured_wrapper import UnstructuredWrapper
4
+ from .base_handler import BaseHandler
5
+
6
+
7
+ class EPUBHandler(BaseHandler):
8
+ extensions = frozenset([".epub"])
9
+
10
+ def __init__(self, *args, **kwargs):
11
+ super().__init__(*args, **kwargs)
12
+ self.unstructured = UnstructuredWrapper()
13
+
14
+ async def handle(self, file_path, *args, **kwargs) -> str:
15
+ """
16
+ Handles EPUB files by converting them to Markdown using Unstructured.
17
+
18
+ Args:
19
+ file_path: Path to the .epub file.
20
+
21
+ Returns:
22
+ Markdown string representing the EPUB content, or an error message.
23
+ """
24
+ logger.info(f"Processing EPUB file {file_path}")
25
+ try:
26
+ md_content = await self.unstructured.convert(file_path) # added await
27
+ if md_content and ensure_minimum_content(md_content):
28
+ return md_content
29
+ else:
30
+ raise RuntimeError(f"EPUB conversion failed or insufficient content: {file_path}")
31
+ except Exception as e:
32
+ logger.error(f"Error handling EPUB file '{file_path}': {e}")
33
+ return "# Error al procesar EPUB"
@@ -0,0 +1,48 @@
1
+ from ..common.logger import logger
2
+ from ..common.utils import ensure_minimum_content
3
+ from ..converters.gpt4o_mini_vision import GPT4oMiniVisionWrapper
4
+ from .base_handler import BaseHandler
5
+
6
+
7
+ class ImageHandler(BaseHandler):
8
+ extensions = frozenset(
9
+ [
10
+ ".bmp",
11
+ ".gif",
12
+ ".heic",
13
+ ".jpeg",
14
+ ".jpg",
15
+ ".png",
16
+ ".prn",
17
+ ".svg",
18
+ ".tiff",
19
+ ".webp",
20
+ ".heif",
21
+ ]
22
+ )
23
+
24
+ def __init__(self, *args, **kwargs):
25
+ super().__init__(*args, **kwargs)
26
+ self.gpt4o_mini_vision = GPT4oMiniVisionWrapper()
27
+
28
+ async def handle(self, file_path, *args, **kwargs) -> str:
29
+ """
30
+ Handles image files by converting them to Markdown using GPT-4o-mini Vision.
31
+
32
+ Args:
33
+ file_path: Path to the image file.
34
+
35
+ Returns:
36
+ Markdown string representing the image content (OCR and analysis),
37
+ or an error message.
38
+ """
39
+ logger.info(f"Processing image file: {file_path}")
40
+ try:
41
+ md_content = await self.gpt4o_mini_vision.convert(file_path)
42
+ if md_content and ensure_minimum_content(md_content):
43
+ return md_content
44
+ else:
45
+ raise RuntimeError(f"Image conversion failed or insufficient content: {file_path}")
46
+ except Exception as e:
47
+ logger.error(f"Error handling image file '{file_path}': {e}")
48
+ return "# Error al procesar imagen"
@@ -0,0 +1,31 @@
1
+ import nbformat
2
+
3
+ from ..common import utils
4
+ from ..common.logger import logger
5
+ from .base_handler import BaseHandler
6
+
7
+
8
+ class IpynbHandler(BaseHandler):
9
+ """Handler for Jupyter notebooks (.ipynb)."""
10
+
11
+ extensions = frozenset([".ipynb"])
12
+
13
+ async def handle(self, file_path: str, *args, **kwargs) -> str | None:
14
+ logger.info(f"Processing notebook: {file_path}")
15
+ try:
16
+ nb = nbformat.read(file_path, as_version=4)
17
+ cells_content = []
18
+ for cell in nb.cells:
19
+ if cell.cell_type == "markdown":
20
+ cells_content.append(cell.source)
21
+ elif cell.cell_type == "code":
22
+ cells_content.append("```python\n" + cell.source + "\n```")
23
+ result = "\n\n".join(cells_content)
24
+
25
+ if utils.ensure_minimum_content(result):
26
+ return result
27
+ else:
28
+ return None
29
+ except Exception as e:
30
+ logger.error(f"Error processing notebook {file_path}: {e}")
31
+ return None
@@ -0,0 +1,75 @@
1
+ import json
2
+ import os
3
+
4
+ import yaml
5
+ from bs4 import BeautifulSoup
6
+
7
+ try:
8
+ import chardet
9
+ except ImportError:
10
+ chardet = None
11
+
12
+ from ..common.logger import logger
13
+ from ..common.utils import ensure_minimum_content
14
+ from .base_handler import BaseHandler
15
+
16
+
17
+ class MarkupHandler(BaseHandler):
18
+ """Handler for .html, .xml, .json, .ndjson, .yaml, .yml files."""
19
+
20
+ extensions = frozenset([".html", ".htm", ".xml", ".json", ".ndjson", ".yaml", ".yml"])
21
+
22
+ async def handle(self, file_path: str, *args, **kwargs) -> str:
23
+ logger.info(f"Processing markup file: {file_path}")
24
+ try:
25
+ ext = os.path.splitext(file_path)[1].lower()
26
+
27
+ # Detect encoding
28
+ encoding = "utf-8" # Default
29
+ if chardet:
30
+ with open(file_path, "rb") as f:
31
+ raw_data = f.read()
32
+ result = chardet.detect(raw_data)
33
+ encoding = result["encoding"]
34
+ logger.debug(f"Detected encoding: {encoding}")
35
+ else:
36
+ logger.warning("chardet not available, assuming UTF-8 encoding.")
37
+
38
+ with open(file_path, "r", encoding=encoding) as f:
39
+ content = f.read()
40
+
41
+ if ext in [".html", ".htm"]:
42
+ soup = BeautifulSoup(content, "html.parser")
43
+ text = soup.get_text(separator="\n")
44
+ elif ext == ".xml":
45
+ soup = BeautifulSoup(content, "xml")
46
+ text = soup.get_text(separator="\n")
47
+ elif ext in [".json", ".ndjson"]:
48
+ try:
49
+ # Attempt to parse as complete JSON
50
+ data = json.loads(content)
51
+ text = json.dumps(data, indent=2, ensure_ascii=False)
52
+ except Exception:
53
+ # If not, process line by line (ndjson case)
54
+ lines = content.splitlines()
55
+ parsed_lines = []
56
+ for line in lines:
57
+ try:
58
+ obj = json.loads(line)
59
+ parsed_lines.append(json.dumps(obj, indent=2, ensure_ascii=False))
60
+ except Exception:
61
+ parsed_lines.append(line)
62
+ text = "\n".join(parsed_lines)
63
+ elif ext in [".yaml", ".yml"]:
64
+ data = yaml.safe_load(content)
65
+ text = yaml.dump(data, allow_unicode=True)
66
+ else:
67
+ text = content
68
+
69
+ if ensure_minimum_content(text):
70
+ return text
71
+ else:
72
+ raise RuntimeError(f"Insufficient content after conversion: {file_path}")
73
+ except Exception as e:
74
+ logger.error(f"Error processing markup file {file_path}: {e}")
75
+ raise
@@ -0,0 +1,47 @@
1
+ from ..common.logger import logger
2
+ from ..common.utils import ensure_minimum_content
3
+ from ..converters.azure_docint import AzureDocIntWrapper
4
+ from ..converters.unstructured_wrapper import UnstructuredWrapper
5
+ from .base_handler import BaseHandler
6
+
7
+
8
+ class OfficeHandler(BaseHandler):
9
+ extensions = frozenset([".doc", ".docx", ".odt", ".rtf", ".ppt", ".pptx"])
10
+
11
+ def __init__(self, *args, **kwargs):
12
+ super().__init__(*args, **kwargs)
13
+ self.azure_docint = AzureDocIntWrapper()
14
+ self.unstructured = UnstructuredWrapper()
15
+
16
+ async def handle(self, file_path, *args, **kwargs) -> str:
17
+ """
18
+ Handles Office documents (.doc, .docx, .odt, .rtf, .ppt, .pptx) by
19
+ first trying Azure Document Intelligence and falling back to Unstructured.
20
+
21
+ Args:
22
+ file_path: Path to the Office document file.
23
+
24
+ Returns:
25
+ Markdown string representing the document content, or an error message.
26
+ """
27
+ logger.info(f"Processing Office document: {file_path}")
28
+ try:
29
+ # First try with Azure Document Intelligence
30
+ logger.info(f"Attempting conversion with Azure Document Intelligence for: {file_path}")
31
+ md_content = await self.azure_docint.convert(file_path)
32
+ if md_content and ensure_minimum_content(md_content):
33
+ return md_content
34
+
35
+ # Fallback to Unstructured if Azure Doc Intelligence fails or returns insufficient content
36
+ logger.info(f"Falling back to Unstructured for: {file_path}")
37
+ md_content = await self.unstructured.convert(file_path)
38
+ if md_content and ensure_minimum_content(md_content):
39
+ return md_content
40
+
41
+ raise RuntimeError(
42
+ f"Office document conversion failed with both Azure Doc Intelligence and Unstructured: {file_path}"
43
+ )
44
+
45
+ except Exception as e:
46
+ logger.error(f"Error handling Office document '{file_path}': {e}")
47
+ return "# Error al procesar documento de Office"
@@ -0,0 +1,122 @@
1
+ from enum import Enum
2
+
3
+ import fitz
4
+
5
+ from ..common.logger import logger
6
+ from ..common.utils import ensure_minimum_content
7
+ from ..converters.azure_docint import AzureDocIntWrapper
8
+ from ..converters.gpt4o_mini_vision import GPT4oMiniVisionWrapper
9
+ from ..converters.markitdown_wrapper import MarkitDownWrapper
10
+ from ..converters.pymupdf_wrapper import PyMuPDFWrapper
11
+ from ..converters.unstructured_wrapper import UnstructuredWrapper
12
+ from .base_handler import BaseHandler
13
+
14
+
15
+ class PDFType(Enum):
16
+ """
17
+ Tipos de PDF detectados por el PDFHandler.
18
+
19
+ - TEXT_ONLY: PDF que contiene solo texto, sin imágenes.
20
+ - TEXT_PLUS_IMAGES: PDF que contiene texto e imágenes.
21
+ - ALL_IMAGES: PDF que contiene solo imágenes (PDF escaneado).
22
+ """
23
+
24
+ TEXT_ONLY = "TEXT_ONLY"
25
+ TEXT_PLUS_IMAGES = "TEXT_PLUS_IMAGES"
26
+ ALL_IMAGES = "ALL_IMAGES"
27
+
28
+
29
+ class PDFHandler(BaseHandler):
30
+ extensions = frozenset([".pdf"])
31
+
32
+ def __init__(self, *args, **kwargs):
33
+ super().__init__(*args, **kwargs)
34
+ self.markitdown = MarkitDownWrapper()
35
+ self.unstructured = UnstructuredWrapper()
36
+ self.azure_docint = AzureDocIntWrapper()
37
+ self.gpt4o_mini_vision = GPT4oMiniVisionWrapper()
38
+ self.pymu = PyMuPDFWrapper()
39
+
40
+ self.text_pipeline = [
41
+ self.markitdown,
42
+ self.unstructured,
43
+ self.pymu,
44
+ self.azure_docint,
45
+ ]
46
+ self.image_pipeline = [
47
+ self.gpt4o_mini_vision,
48
+ ]
49
+
50
+ async def handle(self, file_path, *args, **kwargs):
51
+ try:
52
+ pdf_type = await self._detect_pdf_type(file_path)
53
+
54
+ if pdf_type == PDFType.TEXT_ONLY:
55
+ pipeline = self.text_pipeline
56
+ elif pdf_type == PDFType.ALL_IMAGES:
57
+ pipeline = self.image_pipeline
58
+ elif pdf_type == PDFType.TEXT_PLUS_IMAGES:
59
+ pipeline = self.text_pipeline + self.image_pipeline
60
+ else:
61
+ pipeline = self.text_pipeline
62
+
63
+ for converter in pipeline:
64
+ logger.info(f"Trying {converter.name} for PDF {file_path}")
65
+ try:
66
+ md_content = await converter.convert(file_path)
67
+ if md_content and ensure_minimum_content(md_content):
68
+ return md_content
69
+ except Exception as e:
70
+ logger.error(f"Converter {converter.name} failed for PDF {file_path}: {e}")
71
+
72
+ raise RuntimeError(f"PDF conversion failed with all converters for {file_path}")
73
+
74
+ except Exception as e:
75
+ logger.error(f"Error handling PDF '{file_path}': {e}")
76
+ return None
77
+
78
+ async def _detect_pdf_type(self, file_path: str) -> PDFType:
79
+ """
80
+ Detect the type of PDF file based on its content.
81
+ """
82
+ min_text_length_threshold = 50
83
+
84
+ try:
85
+
86
+ async def open_and_process_doc():
87
+ with fitz.open(file_path) as doc:
88
+ total_pages = doc.page_count
89
+ pages_with_text = 0
90
+ pages_with_images = 0
91
+
92
+ for page_index in range(total_pages):
93
+ page = doc.load_page(page_index)
94
+ page_text = page.get_text().strip()
95
+ if len(page_text) >= min_text_length_threshold:
96
+ pages_with_text += 1
97
+ images = page.get_images(full=True)
98
+ if images:
99
+ pages_with_images += 1
100
+
101
+ logger.debug(f"Pages with text: {pages_with_text}/{total_pages}")
102
+ logger.debug(f"Pages with images: {pages_with_images}/{total_pages}")
103
+
104
+ is_text_only = pages_with_text == total_pages and pages_with_images == 0
105
+ is_all_images = pages_with_images == total_pages and pages_with_text == 0
106
+ has_text_and_images = pages_with_text > 0 and pages_with_images > 0
107
+
108
+ if is_text_only:
109
+ return PDFType.TEXT_ONLY
110
+ elif is_all_images:
111
+ return PDFType.ALL_IMAGES
112
+ elif has_text_and_images:
113
+ return PDFType.TEXT_PLUS_IMAGES
114
+ else:
115
+ return PDFType.TEXT_PLUS_IMAGES # Fallback
116
+
117
+ return await open_and_process_doc() # Run in thread
118
+ except Exception as e:
119
+ logger.error(f"Error analyzing PDF '{file_path}': {e}")
120
+ # Important to re-raise the exception after logging, so the
121
+ # caller knows something went wrong *during the analysis*.
122
+ raise
@@ -0,0 +1,153 @@
1
+ import os
2
+ from typing import Optional
3
+
4
+ from .base_handler import BaseHandler
5
+
6
+ try:
7
+ from libratom.lib.pff import PffArchive
8
+
9
+ HAS_LIBRATOM = True
10
+ except ImportError:
11
+ HAS_LIBRATOM = False
12
+
13
+ from ..common.logger import logger
14
+ from ..common.utils import clean_markdown, ensure_minimum_content
15
+
16
+
17
+ class PSTHandler(BaseHandler):
18
+ extensions = frozenset([".pst"])
19
+
20
+ async def handle(self, file_path, *args, **kwargs) -> str:
21
+ """
22
+ Parses a PST file using libratom, extracts messages and attachments,
23
+ and converts the content to Markdown. Recursively processes attachments.
24
+
25
+ Args:
26
+ file_path: Path to the .pst file.
27
+
28
+ Returns:
29
+ Markdown string representing the PST content, or an error message.
30
+ """
31
+ if not HAS_LIBRATOM:
32
+ logger.error("libratom is not installed. PST processing is disabled.")
33
+ return "# Error: libratom not installed. Cannot process PST files."
34
+
35
+ logger.info(f"Processing PST file: {file_path}")
36
+ try:
37
+ markdown_content = self._process_pst(file_path)
38
+ if markdown_content:
39
+ return markdown_content
40
+ else:
41
+ return "# PST Archive\n\n(No messages found or insufficient content.)"
42
+ except Exception as e:
43
+ logger.error(f"Error processing PST file: {file_path}: {e}")
44
+ return f"# Error processing PST file: {e}"
45
+
46
+ def _process_pst(self, file_path: str) -> Optional[str]:
47
+ """
48
+ Parses the PST file, extracts messages and attachments, and converts to Markdown.
49
+
50
+ Args:
51
+ file_path: Path to the PST file.
52
+
53
+ Returns:
54
+ Markdown string, or None if no messages are found or content is insufficient.
55
+ """
56
+ if not os.path.isfile(file_path):
57
+ logger.error(f"PST file not found: {file_path}")
58
+ return None
59
+
60
+ all_md_parts = [f"# PST Archive: {os.path.basename(file_path)}\n"]
61
+
62
+ try:
63
+ with PffArchive(file_path) as archive:
64
+ for folder in archive.folders():
65
+ if not folder.name: # Skip folders with no name
66
+ continue
67
+
68
+ # Add folder information, handling None folder names
69
+ all_md_parts.append(f"\n## Folder: {folder.name or '(Unnamed Folder)'}\n")
70
+ message_count = 0
71
+
72
+ for message in folder.messages():
73
+ message_count += 1
74
+ try:
75
+ message_md = self._process_message(message, message_count)
76
+ if message_md:
77
+ all_md_parts.extend(message_md)
78
+ except Exception as e:
79
+ logger.error(
80
+ f"Error processing message {message_count} in folder {folder.name}: {e}"
81
+ )
82
+ all_md_parts.append(
83
+ f"### Error processing message {message_count}: {e}"
84
+ )
85
+
86
+ final_md = clean_markdown("\n\n".join(all_md_parts))
87
+ return final_md if ensure_minimum_content(final_md) else None
88
+
89
+ except Exception as e:
90
+ logger.error(f"Error opening or processing PST archive {file_path}: {e}")
91
+ return None
92
+
93
+ def _process_message(self, message, message_count: int) -> Optional[list[str]]:
94
+ """
95
+ Processes a single message from the PST archive.
96
+
97
+ Args:
98
+ message: The message object from libratom.
99
+ message_count: The message number within the folder (for display).
100
+
101
+ Returns:
102
+ A list of Markdown strings representing the message, or None on error.
103
+ """
104
+ try:
105
+ subject = message.subject or "(No Subject)"
106
+ sender = "Unknown Sender"
107
+ date_ = "Unknown Date"
108
+
109
+ # Extract headers safely, handling potential errors
110
+ try:
111
+ headers = message.transport_headers
112
+ if headers:
113
+ if isinstance(headers, bytes):
114
+ headers = headers.decode(errors="replace")
115
+ for line in headers.splitlines():
116
+ if line.lower().startswith("from:"):
117
+ sender = line.split(":", 1)[1].strip()
118
+ elif line.lower().startswith("date:"):
119
+ date_ = line.split(":", 1)[1].strip()
120
+ except Exception as header_err:
121
+ logger.warning(f"Error parsing headers: {header_err}")
122
+
123
+ # Extract the body, handling different encodings and body types
124
+ body_content = ""
125
+ try:
126
+ if message.plain_text_body:
127
+ body_content = message.plain_text_body.decode(errors="replace")
128
+ elif message.html_body:
129
+ body_content = message.html_body.decode(errors="replace")
130
+ elif message.rtf_body:
131
+ body_content = message.rtf_body.decode(errors="replace")
132
+
133
+ except Exception as body_err:
134
+ logger.warning(f"Error decoding message body: {body_err}")
135
+
136
+ message_md_parts = [
137
+ f"### Message {message_count}",
138
+ f"**Subject:** {subject}",
139
+ f"**From:** {sender}",
140
+ f"**Date:** {date_}",
141
+ "",
142
+ "```",
143
+ body_content.strip() if body_content else "[No body text]",
144
+ "```",
145
+ ]
146
+
147
+ # Handle attachments todo: handle attachments
148
+
149
+ return message_md_parts
150
+
151
+ except Exception as e:
152
+ logger.error(f"Error processing individual message: {e}")
153
+ return None