markitdown-pro 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. markitdown_pro/__init__.py +1 -0
  2. markitdown_pro/common/__init__.py +0 -0
  3. markitdown_pro/common/logger.py +6 -0
  4. markitdown_pro/common/utils.py +59 -0
  5. markitdown_pro/conversion_pipeline.py +212 -0
  6. markitdown_pro/converters/__init__.py +0 -0
  7. markitdown_pro/converters/azure_docint.py +13 -0
  8. markitdown_pro/converters/base.py +26 -0
  9. markitdown_pro/converters/gpt4o_mini_vision.py +21 -0
  10. markitdown_pro/converters/markitdown_wrapper.py +44 -0
  11. markitdown_pro/converters/pymupdf_wrapper.py +42 -0
  12. markitdown_pro/converters/unstructured_wrapper.py +62 -0
  13. markitdown_pro/converters/youtube_wrapper.py +67 -0
  14. markitdown_pro/handlers/__init__.py +0 -0
  15. markitdown_pro/handlers/audio_handler.py +40 -0
  16. markitdown_pro/handlers/base_handler.py +16 -0
  17. markitdown_pro/handlers/email_handler.py +169 -0
  18. markitdown_pro/handlers/epub_handler.py +33 -0
  19. markitdown_pro/handlers/image_handler.py +48 -0
  20. markitdown_pro/handlers/ipynb_handler.py +31 -0
  21. markitdown_pro/handlers/markup_handler.py +75 -0
  22. markitdown_pro/handlers/office_handler.py +47 -0
  23. markitdown_pro/handlers/pdf_handler.py +122 -0
  24. markitdown_pro/handlers/pst_handler.py +153 -0
  25. markitdown_pro/handlers/tabular_handler.py +34 -0
  26. markitdown_pro/handlers/text_handler.py +38 -0
  27. markitdown_pro/services/__init__.py +0 -0
  28. markitdown_pro/services/azure_service.py +160 -0
  29. markitdown_pro/services/openai_services.py +209 -0
  30. markitdown_pro-0.1.0.dist-info/METADATA +367 -0
  31. markitdown_pro-0.1.0.dist-info/RECORD +33 -0
  32. markitdown_pro-0.1.0.dist-info/WHEEL +5 -0
  33. markitdown_pro-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,34 @@
1
+ import os
2
+
3
+ import pandas as pd
4
+
5
+ from ..common.logger import logger
6
+ from ..common.utils import ensure_minimum_content
7
+ from .base_handler import BaseHandler
8
+
9
+
10
+ class TabularHandler(BaseHandler):
11
+ """Handler for .csv, .tsv, .xls, .xlsx files."""
12
+
13
+ extensions = frozenset([".csv", ".tsv", ".xls", ".xlsx"])
14
+
15
+ async def handle(self, file_path: str, *args, **kwargs) -> str:
16
+ logger.info(f"Processing tabular file: {file_path}")
17
+ try:
18
+ ext = os.path.splitext(file_path)[1].lower()
19
+ if ext in [".csv", ".tsv"]:
20
+ delimiter = "\t" if ext == ".tsv" else ","
21
+ df = pd.read_csv(file_path, delimiter=delimiter)
22
+ elif ext in [".xls", ".xlsx"]:
23
+ df = pd.read_excel(file_path)
24
+ else:
25
+ raise RuntimeError("Unsupported tabular format")
26
+ md = df.to_markdown(index=False)
27
+
28
+ if ensure_minimum_content(md):
29
+ return md
30
+ else:
31
+ raise RuntimeError(f"Insufficient content after conversion: {file_path}")
32
+ except Exception as e:
33
+ logger.error(f"Error processing tabular file {file_path}: {e}")
34
+ raise
@@ -0,0 +1,38 @@
1
+ import os
2
+
3
+ import chardet
4
+
5
+ from ..common.logger import logger
6
+ from ..common.utils import clean_markdown, ensure_minimum_content
7
+ from .base_handler import BaseHandler
8
+
9
+
10
+ class TextHandler(BaseHandler):
11
+ """Handler for .txt, .md, .py, .go, and other text/code files."""
12
+
13
+ extensions = frozenset([".txt", ".md", ".py", ".go"])
14
+
15
+ async def handle(self, file_path: str, *args, **kwargs) -> str:
16
+ logger.info(f"Processing text file: {file_path}")
17
+ try:
18
+ with open(file_path, "rb") as f:
19
+ raw_data = f.read()
20
+ result = chardet.detect(raw_data)
21
+ encoding = result["encoding"] or "utf-8"
22
+ logger.debug(f"Detected encoding: {encoding}")
23
+ content = raw_data.decode(encoding, errors="replace")
24
+
25
+ ext = os.path.splitext(file_path)[1].lower()
26
+ if ext == ".md":
27
+ content = clean_markdown(content)
28
+
29
+ if ensure_minimum_content(content):
30
+ return content
31
+ else:
32
+ raise RuntimeError(
33
+ f"Insufficient content. Content length is {len(content)} characters."
34
+ )
35
+ except Exception as e:
36
+ file_name = os.path.basename(file_path)
37
+ logger.error(f"Error processing text file {file_name}: {e}")
38
+ raise
File without changes
@@ -0,0 +1,160 @@
1
+ import logging
2
+ import os
3
+ from pathlib import Path
4
+ from typing import Optional
5
+
6
+ import azure.cognitiveservices.speech as speechsdk
7
+ from azure.ai.documentintelligence import DocumentIntelligenceClient
8
+ from azure.ai.documentintelligence.models import DocumentAnalysisFeature
9
+ from azure.ai.documentintelligence.models import DocumentContentFormat as ContentFormat
10
+ from azure.core.credentials import AzureKeyCredential
11
+
12
+ from ..common.logger import logger
13
+ from ..common.utils import clean_markdown, ensure_minimum_content
14
+
15
+
16
+ class AzureServices:
17
+ def __init__(self=None):
18
+ self.azure_doc_client = None
19
+ self.azure_speech_config = None
20
+ self.azure_speech_voice_name = "en-US-AndrewMultilingualNeural"
21
+ self._initialize_azure_services()
22
+
23
+ def _initialize_azure_services(self):
24
+
25
+ # Initialize Azure Document Intelligence client
26
+ azure_docint_endpoint = os.getenv("AZURE_DOCINTEL_ENDPOINT", "")
27
+ azure_docint_key = os.getenv("AZURE_DOCINTEL_KEY", "")
28
+ if azure_docint_endpoint and azure_docint_key:
29
+ try:
30
+ self.azure_doc_client = DocumentIntelligenceClient(
31
+ endpoint=azure_docint_endpoint,
32
+ credential=AzureKeyCredential(azure_docint_key),
33
+ )
34
+ except Exception as e:
35
+ logging.warning(f"Failed to initialize Azure Document Intelligence client: {e}")
36
+
37
+ azure_speech_key = os.getenv("AZURE_SPEECH_KEY", "")
38
+ azure_speech_region = os.getenv("AZURE_SPEECH_REGION", "")
39
+ if azure_speech_key and azure_speech_region:
40
+ try:
41
+ self.azure_speech_config = speechsdk.SpeechConfig(
42
+ subscription=azure_speech_key, region=azure_speech_region
43
+ )
44
+ except Exception as e:
45
+ logging.warning(f"Failed to initialize Azure Speech Service configuration: {e}")
46
+
47
+ def process_azure_doc_intelligence(self, file_path: str) -> Optional[str]:
48
+ """
49
+ Convert a document to Markdown using Azure Document Intelligence.
50
+ """
51
+ if not self.azure_doc_client:
52
+ logger.info("Azure Document Intelligence client not configured, skipping conversion.")
53
+ return None
54
+
55
+ logger.info(f"Attempting Azure Document Intelligence conversion on '{file_path}'")
56
+ try:
57
+ with open(file_path, "rb") as f:
58
+ file_data = f.read()
59
+
60
+ # Detect file extension
61
+ extension = Path(file_path).suffix.lower()
62
+ if extension not in [".docx", ".xlsx", ".pptx"]:
63
+ features = [DocumentAnalysisFeature.LANGUAGES]
64
+ else:
65
+ features = []
66
+
67
+ poller = self.azure_doc_client.begin_analyze_document(
68
+ model_id="prebuilt-layout",
69
+ body=file_data,
70
+ content_type="application/octet-stream",
71
+ features=features,
72
+ output_content_format=ContentFormat.MARKDOWN,
73
+ )
74
+ result = poller.result(timeout=120)
75
+
76
+ if hasattr(result, "content"):
77
+ content = result.content or ""
78
+
79
+ else:
80
+ # Fallback for older API responses
81
+ lines = []
82
+ for page in result.pages:
83
+ for line in page.lines:
84
+ lines.append(line.content)
85
+ content = "\n".join(lines)
86
+
87
+ final_md = clean_markdown(content)
88
+ if ensure_minimum_content(final_md):
89
+ return final_md
90
+
91
+ logger.info("Azure Document Intelligence conversion returned insufficient content.")
92
+ return None
93
+
94
+ except Exception as e:
95
+ logger.error(f"Error during Azure Document Intelligence conversion: {e}")
96
+ return None
97
+
98
+ async def recognize_azure_speech_to_text_from_file(self, file_path: str) -> Optional[str]:
99
+ """
100
+ Recognize speech from an audio file with automatic language detection
101
+ across the top 6 spoken languages globally.
102
+
103
+ Args:
104
+ file_path (str): Path to the audio file.
105
+
106
+ Returns:
107
+ Optional[str]: Transcribed text if successful, None otherwise.
108
+ """
109
+ if not self.azure_speech_config:
110
+ logging.info("Azure Speech Service not configured, skipping speech recognition.")
111
+ return None
112
+
113
+ try:
114
+ audio_config = speechsdk.AudioConfig(filename=file_path)
115
+ languages = ["en-US", "zh-CN", "hi-IN", "es-ES"]
116
+
117
+ # Configure auto language detection with the specified languages
118
+ auto_detect_source_language_config = (
119
+ speechsdk.languageconfig.AutoDetectSourceLanguageConfig(languages=languages)
120
+ )
121
+
122
+ # Create a speech recognizer with the auto language detection configuration
123
+ speech_recognizer = speechsdk.SpeechRecognizer(
124
+ speech_config=self.azure_speech_config,
125
+ audio_config=audio_config,
126
+ auto_detect_source_language_config=auto_detect_source_language_config,
127
+ )
128
+
129
+ # Perform speech recognition
130
+ result = await speech_recognizer.recognize_once_async()
131
+
132
+ # Check the result
133
+ if result.reason == speechsdk.ResultReason.RecognizedSpeech:
134
+ # Retrieve the detected language
135
+ detected_language = result.properties.get(
136
+ speechsdk.PropertyId.SpeechServiceConnection_AutoDetectSourceLanguageResult,
137
+ "Unknown",
138
+ )
139
+ logging.debug(f"Detected Language {detected_language}")
140
+ return result.text
141
+
142
+ elif result.reason == speechsdk.ResultReason.NoMatch:
143
+ logging.warning("No speech could be recognized from the audio.")
144
+ return None
145
+
146
+ elif result.reason == speechsdk.ResultReason.Canceled:
147
+ cancellation_details = speechsdk.CancellationDetails(result)
148
+ logging.error(
149
+ f"Speech Recognition canceled: {cancellation_details.reason}. "
150
+ f"Error details: {cancellation_details.error_details}"
151
+ )
152
+ return None
153
+
154
+ else:
155
+ logging.error("Unknown error occurred during speech recognition.")
156
+ return None
157
+
158
+ except Exception as e:
159
+ logging.error(f"An error occurred during speech recognition: {e}")
160
+ return None
@@ -0,0 +1,209 @@
1
+ import asyncio
2
+ import base64
3
+
4
+ # import concurrent.futures
5
+ import mimetypes
6
+ import os
7
+ import re
8
+ import tempfile
9
+ from pathlib import Path
10
+ from typing import Optional, Tuple
11
+
12
+ import fitz
13
+ from langchain_core.messages import HumanMessage
14
+ from langchain_openai import AzureChatOpenAI
15
+ from PIL import Image as PILImage
16
+
17
+ from ..common.logger import logger
18
+ from ..common.utils import clean_markdown, ensure_minimum_content
19
+
20
+
21
+ class GPT4oMiniVision:
22
+ def __init__(self=None):
23
+ azure_key = os.getenv("AZURE_OPENAI_API_KEY")
24
+ api_version = os.getenv("AZURE_OPENAI_API_VERSION")
25
+
26
+ GPT4OMINI_DEPLOYMENT_NAME: str = os.getenv("GPT4oMINI_DEPLOYMENT_NAME", "")
27
+ COMPLETION_TOKENS: int = 4096
28
+
29
+ self.client = None
30
+ if GPT4OMINI_DEPLOYMENT_NAME:
31
+ try:
32
+ from pydantic import BaseModel
33
+
34
+ class ImageSchema(BaseModel):
35
+ ocr: str
36
+ analysis: str
37
+
38
+ llm = AzureChatOpenAI(
39
+ deployment_name=GPT4OMINI_DEPLOYMENT_NAME,
40
+ temperature=0,
41
+ max_tokens=COMPLETION_TOKENS,
42
+ streaming=False,
43
+ api_version=api_version,
44
+ api_key=azure_key,
45
+ cache=False,
46
+ )
47
+ self.client = llm.with_structured_output(ImageSchema)
48
+ logger.info("GPT-4o-mini client initialized successfully.")
49
+ except Exception as e:
50
+ logger.warning(f"Failed to initialize GPT-4o-mini client: {e}")
51
+ self.client = None
52
+
53
+ def _build_image_url_block(self, file_or_url: str) -> dict:
54
+ if isinstance(file_or_url, Path):
55
+ file_or_url = str(file_or_url)
56
+
57
+ # If the input is a URL, return it as an image_url block
58
+ if re.match(r"^https?://", file_or_url, re.IGNORECASE):
59
+ return {
60
+ "type": "image_url",
61
+ "image_url": {"url": file_or_url},
62
+ }
63
+
64
+ path = Path(file_or_url)
65
+ if not path.is_file():
66
+ raise ValueError(f"Local file not found: {file_or_url}")
67
+
68
+ try:
69
+ content_type, _ = mimetypes.guess_type(file_or_url)
70
+ if not content_type:
71
+ content_type = "image/jpeg"
72
+ logger.warning(
73
+ f"Could not determine content type for {file_or_url}, defaulting to image/jpeg"
74
+ )
75
+
76
+ # Convert the image to JPEG and encode it in Base64
77
+ img = PILImage.open(file_or_url)
78
+ with tempfile.NamedTemporaryFile(suffix=".jpg", delete=False) as tmp_file:
79
+ img.convert("RGB").save(tmp_file.name, "JPEG")
80
+ jpeg_path = tmp_file.name
81
+
82
+ with open(jpeg_path, "rb") as f:
83
+ b64_data = base64.b64encode(f.read()).decode("utf-8")
84
+
85
+ os.unlink(jpeg_path) # Delete the temporary JPEG file
86
+
87
+ return {
88
+ "type": "image_url",
89
+ "image_url": {"url": f"data:{content_type};base64,{b64_data}"},
90
+ }
91
+ except Exception as e:
92
+ logger.error(f"Error building image block for {file_or_url}: {e}")
93
+ raise
94
+
95
+ async def process_image(self, file_or_url: str, prompt: Optional[str] = None) -> Optional[str]:
96
+ if not prompt:
97
+ prompt = (
98
+ "Perform accurate OCR: answer with the markdown of the content of the image. Visually appealing markdown, nothing else. "
99
+ "Perform Image Analysis: answer with a two line image analysis explaining what you see."
100
+ )
101
+
102
+ try:
103
+ image_block = self._build_image_url_block(file_or_url)
104
+ message_content = [
105
+ {"type": "text", "text": prompt},
106
+ image_block,
107
+ ]
108
+ message = HumanMessage(content=message_content)
109
+ response = self.client.invoke([message])
110
+ md_content = clean_markdown(response.analysis)
111
+ md_content += "\n\n" + clean_markdown(response.ocr)
112
+
113
+ return md_content if ensure_minimum_content(md_content) else None
114
+ except Exception as e:
115
+ logger.error(f"process_image: Error during GPT-4o-mini image OCR: {e}")
116
+ return None
117
+
118
+ async def process_scanned_pdf_concurrent(self, file_path: str) -> Optional[str]:
119
+ if not self.client:
120
+ return None
121
+
122
+ try:
123
+ doc = fitz.open(file_path)
124
+ num_pages = doc.page_count
125
+ results = []
126
+
127
+ async def ocr_page(page_index: int) -> Tuple[int, str]:
128
+ try:
129
+ page = doc.load_page(page_index)
130
+ pix = page.get_pixmap(dpi=150)
131
+
132
+ fd, png_path = tempfile.mkstemp(prefix=f"{page_index:3}-", suffix=".png")
133
+ os.close(fd)
134
+ pix.save(png_path)
135
+
136
+ try:
137
+ file_name = os.path.basename(file_path)
138
+ partial_md = await self.process_image(png_path)
139
+ if partial_md:
140
+ logger.debug(
141
+ f"ocr_page: {file_name} - page {page_index}: returned {len(partial_md)} characters"
142
+ )
143
+ else:
144
+ logger.warning(
145
+ f"ocr_page: {file_name} - page {page_index}: no OCR result"
146
+ )
147
+ partial_md = "[No OCR result]"
148
+ return (page_index, f"## Page {page_index + 1}\n\n{partial_md}")
149
+ finally:
150
+ try:
151
+ if os.path.exists(png_path):
152
+ # Remove the temporary PNG file after processing
153
+ logger.debug(f"Removing temporary PNG file: {png_path}")
154
+ Path(png_path).unlink(missing_ok=True)
155
+ except Exception as e:
156
+ logger.error(f"ocr_page: Error removing temporary PNG file: {e}")
157
+ except Exception as e:
158
+ logger.error(f"ocr_page: Error processing page {page_index}: {e}")
159
+ return (page_index, None)
160
+
161
+ logger.info(
162
+ f"process_scanned_pdf_concurrent: Starting concurrent GPT-4o-mini OCR on PDF: '{file_path}'"
163
+ )
164
+
165
+ # run tasks concurrently
166
+ tasks = [ocr_page(i) for i in range(num_pages)]
167
+ results = await asyncio.gather(*tasks)
168
+
169
+ combined_md = clean_markdown("\n\n".join(block for _, block in results))
170
+ return combined_md if ensure_minimum_content(combined_md) else None
171
+
172
+ except Exception as e:
173
+ logger.error(f"Concurrent GPT-4o-mini OCR error for PDF {file_path}: {e}")
174
+ return None
175
+
176
+ async def process_scanned_pdf_simple(self, file_path: str) -> Optional[str]:
177
+ if not self.client:
178
+ return None
179
+
180
+ try:
181
+ doc = fitz.open(file_path)
182
+ all_md_blocks = []
183
+
184
+ logger.info(f"Starting simple GPT-4o-mini OCR on PDF: '{file_path}'")
185
+ for i in range(doc.page_count):
186
+ page = doc.load_page(i)
187
+ pix = page.get_pixmap(dpi=150)
188
+ with tempfile.NamedTemporaryFile(suffix=".png", delete=False) as tmp_file:
189
+ png_path = tmp_file.name
190
+ pix.save(png_path)
191
+ try:
192
+ partial_md = await self.process_image(png_path)
193
+ if partial_md:
194
+ block = f"## Page {i + 1}\n\n{partial_md}"
195
+ else:
196
+ block = f"## Page {i + 1}\n\n[No OCR result]"
197
+ all_md_blocks.append(block)
198
+ finally:
199
+ try:
200
+ Path(png_path).unlink(missing_ok=True)
201
+ except Exception as e:
202
+ logger.error(f"Error removing temporary PNG file in simple OCR: {e}")
203
+
204
+ combined_md = clean_markdown("\n\n".join(all_md_blocks))
205
+ return combined_md if ensure_minimum_content(combined_md) else None
206
+
207
+ except Exception as e:
208
+ logger.error(f"Simple GPT-4o-mini OCR error for PDF {file_path}: {e}")
209
+ return None