markitdown-pro 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- markitdown_pro/__init__.py +1 -0
- markitdown_pro/common/__init__.py +0 -0
- markitdown_pro/common/logger.py +6 -0
- markitdown_pro/common/utils.py +59 -0
- markitdown_pro/conversion_pipeline.py +212 -0
- markitdown_pro/converters/__init__.py +0 -0
- markitdown_pro/converters/azure_docint.py +13 -0
- markitdown_pro/converters/base.py +26 -0
- markitdown_pro/converters/gpt4o_mini_vision.py +21 -0
- markitdown_pro/converters/markitdown_wrapper.py +44 -0
- markitdown_pro/converters/pymupdf_wrapper.py +42 -0
- markitdown_pro/converters/unstructured_wrapper.py +62 -0
- markitdown_pro/converters/youtube_wrapper.py +67 -0
- markitdown_pro/handlers/__init__.py +0 -0
- markitdown_pro/handlers/audio_handler.py +40 -0
- markitdown_pro/handlers/base_handler.py +16 -0
- markitdown_pro/handlers/email_handler.py +169 -0
- markitdown_pro/handlers/epub_handler.py +33 -0
- markitdown_pro/handlers/image_handler.py +48 -0
- markitdown_pro/handlers/ipynb_handler.py +31 -0
- markitdown_pro/handlers/markup_handler.py +75 -0
- markitdown_pro/handlers/office_handler.py +47 -0
- markitdown_pro/handlers/pdf_handler.py +122 -0
- markitdown_pro/handlers/pst_handler.py +153 -0
- markitdown_pro/handlers/tabular_handler.py +34 -0
- markitdown_pro/handlers/text_handler.py +38 -0
- markitdown_pro/services/__init__.py +0 -0
- markitdown_pro/services/azure_service.py +160 -0
- markitdown_pro/services/openai_services.py +209 -0
- markitdown_pro-0.1.0.dist-info/METADATA +367 -0
- markitdown_pro-0.1.0.dist-info/RECORD +33 -0
- markitdown_pro-0.1.0.dist-info/WHEEL +5 -0
- markitdown_pro-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
import os
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
from ..common.logger import logger
|
|
6
|
+
from ..common.utils import ensure_minimum_content
|
|
7
|
+
from .base_handler import BaseHandler
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class TabularHandler(BaseHandler):
|
|
11
|
+
"""Handler for .csv, .tsv, .xls, .xlsx files."""
|
|
12
|
+
|
|
13
|
+
extensions = frozenset([".csv", ".tsv", ".xls", ".xlsx"])
|
|
14
|
+
|
|
15
|
+
async def handle(self, file_path: str, *args, **kwargs) -> str:
|
|
16
|
+
logger.info(f"Processing tabular file: {file_path}")
|
|
17
|
+
try:
|
|
18
|
+
ext = os.path.splitext(file_path)[1].lower()
|
|
19
|
+
if ext in [".csv", ".tsv"]:
|
|
20
|
+
delimiter = "\t" if ext == ".tsv" else ","
|
|
21
|
+
df = pd.read_csv(file_path, delimiter=delimiter)
|
|
22
|
+
elif ext in [".xls", ".xlsx"]:
|
|
23
|
+
df = pd.read_excel(file_path)
|
|
24
|
+
else:
|
|
25
|
+
raise RuntimeError("Unsupported tabular format")
|
|
26
|
+
md = df.to_markdown(index=False)
|
|
27
|
+
|
|
28
|
+
if ensure_minimum_content(md):
|
|
29
|
+
return md
|
|
30
|
+
else:
|
|
31
|
+
raise RuntimeError(f"Insufficient content after conversion: {file_path}")
|
|
32
|
+
except Exception as e:
|
|
33
|
+
logger.error(f"Error processing tabular file {file_path}: {e}")
|
|
34
|
+
raise
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
import os
|
|
2
|
+
|
|
3
|
+
import chardet
|
|
4
|
+
|
|
5
|
+
from ..common.logger import logger
|
|
6
|
+
from ..common.utils import clean_markdown, ensure_minimum_content
|
|
7
|
+
from .base_handler import BaseHandler
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class TextHandler(BaseHandler):
|
|
11
|
+
"""Handler for .txt, .md, .py, .go, and other text/code files."""
|
|
12
|
+
|
|
13
|
+
extensions = frozenset([".txt", ".md", ".py", ".go"])
|
|
14
|
+
|
|
15
|
+
async def handle(self, file_path: str, *args, **kwargs) -> str:
|
|
16
|
+
logger.info(f"Processing text file: {file_path}")
|
|
17
|
+
try:
|
|
18
|
+
with open(file_path, "rb") as f:
|
|
19
|
+
raw_data = f.read()
|
|
20
|
+
result = chardet.detect(raw_data)
|
|
21
|
+
encoding = result["encoding"] or "utf-8"
|
|
22
|
+
logger.debug(f"Detected encoding: {encoding}")
|
|
23
|
+
content = raw_data.decode(encoding, errors="replace")
|
|
24
|
+
|
|
25
|
+
ext = os.path.splitext(file_path)[1].lower()
|
|
26
|
+
if ext == ".md":
|
|
27
|
+
content = clean_markdown(content)
|
|
28
|
+
|
|
29
|
+
if ensure_minimum_content(content):
|
|
30
|
+
return content
|
|
31
|
+
else:
|
|
32
|
+
raise RuntimeError(
|
|
33
|
+
f"Insufficient content. Content length is {len(content)} characters."
|
|
34
|
+
)
|
|
35
|
+
except Exception as e:
|
|
36
|
+
file_name = os.path.basename(file_path)
|
|
37
|
+
logger.error(f"Error processing text file {file_name}: {e}")
|
|
38
|
+
raise
|
|
File without changes
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import os
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Optional
|
|
5
|
+
|
|
6
|
+
import azure.cognitiveservices.speech as speechsdk
|
|
7
|
+
from azure.ai.documentintelligence import DocumentIntelligenceClient
|
|
8
|
+
from azure.ai.documentintelligence.models import DocumentAnalysisFeature
|
|
9
|
+
from azure.ai.documentintelligence.models import DocumentContentFormat as ContentFormat
|
|
10
|
+
from azure.core.credentials import AzureKeyCredential
|
|
11
|
+
|
|
12
|
+
from ..common.logger import logger
|
|
13
|
+
from ..common.utils import clean_markdown, ensure_minimum_content
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class AzureServices:
|
|
17
|
+
def __init__(self=None):
|
|
18
|
+
self.azure_doc_client = None
|
|
19
|
+
self.azure_speech_config = None
|
|
20
|
+
self.azure_speech_voice_name = "en-US-AndrewMultilingualNeural"
|
|
21
|
+
self._initialize_azure_services()
|
|
22
|
+
|
|
23
|
+
def _initialize_azure_services(self):
|
|
24
|
+
|
|
25
|
+
# Initialize Azure Document Intelligence client
|
|
26
|
+
azure_docint_endpoint = os.getenv("AZURE_DOCINTEL_ENDPOINT", "")
|
|
27
|
+
azure_docint_key = os.getenv("AZURE_DOCINTEL_KEY", "")
|
|
28
|
+
if azure_docint_endpoint and azure_docint_key:
|
|
29
|
+
try:
|
|
30
|
+
self.azure_doc_client = DocumentIntelligenceClient(
|
|
31
|
+
endpoint=azure_docint_endpoint,
|
|
32
|
+
credential=AzureKeyCredential(azure_docint_key),
|
|
33
|
+
)
|
|
34
|
+
except Exception as e:
|
|
35
|
+
logging.warning(f"Failed to initialize Azure Document Intelligence client: {e}")
|
|
36
|
+
|
|
37
|
+
azure_speech_key = os.getenv("AZURE_SPEECH_KEY", "")
|
|
38
|
+
azure_speech_region = os.getenv("AZURE_SPEECH_REGION", "")
|
|
39
|
+
if azure_speech_key and azure_speech_region:
|
|
40
|
+
try:
|
|
41
|
+
self.azure_speech_config = speechsdk.SpeechConfig(
|
|
42
|
+
subscription=azure_speech_key, region=azure_speech_region
|
|
43
|
+
)
|
|
44
|
+
except Exception as e:
|
|
45
|
+
logging.warning(f"Failed to initialize Azure Speech Service configuration: {e}")
|
|
46
|
+
|
|
47
|
+
def process_azure_doc_intelligence(self, file_path: str) -> Optional[str]:
|
|
48
|
+
"""
|
|
49
|
+
Convert a document to Markdown using Azure Document Intelligence.
|
|
50
|
+
"""
|
|
51
|
+
if not self.azure_doc_client:
|
|
52
|
+
logger.info("Azure Document Intelligence client not configured, skipping conversion.")
|
|
53
|
+
return None
|
|
54
|
+
|
|
55
|
+
logger.info(f"Attempting Azure Document Intelligence conversion on '{file_path}'")
|
|
56
|
+
try:
|
|
57
|
+
with open(file_path, "rb") as f:
|
|
58
|
+
file_data = f.read()
|
|
59
|
+
|
|
60
|
+
# Detect file extension
|
|
61
|
+
extension = Path(file_path).suffix.lower()
|
|
62
|
+
if extension not in [".docx", ".xlsx", ".pptx"]:
|
|
63
|
+
features = [DocumentAnalysisFeature.LANGUAGES]
|
|
64
|
+
else:
|
|
65
|
+
features = []
|
|
66
|
+
|
|
67
|
+
poller = self.azure_doc_client.begin_analyze_document(
|
|
68
|
+
model_id="prebuilt-layout",
|
|
69
|
+
body=file_data,
|
|
70
|
+
content_type="application/octet-stream",
|
|
71
|
+
features=features,
|
|
72
|
+
output_content_format=ContentFormat.MARKDOWN,
|
|
73
|
+
)
|
|
74
|
+
result = poller.result(timeout=120)
|
|
75
|
+
|
|
76
|
+
if hasattr(result, "content"):
|
|
77
|
+
content = result.content or ""
|
|
78
|
+
|
|
79
|
+
else:
|
|
80
|
+
# Fallback for older API responses
|
|
81
|
+
lines = []
|
|
82
|
+
for page in result.pages:
|
|
83
|
+
for line in page.lines:
|
|
84
|
+
lines.append(line.content)
|
|
85
|
+
content = "\n".join(lines)
|
|
86
|
+
|
|
87
|
+
final_md = clean_markdown(content)
|
|
88
|
+
if ensure_minimum_content(final_md):
|
|
89
|
+
return final_md
|
|
90
|
+
|
|
91
|
+
logger.info("Azure Document Intelligence conversion returned insufficient content.")
|
|
92
|
+
return None
|
|
93
|
+
|
|
94
|
+
except Exception as e:
|
|
95
|
+
logger.error(f"Error during Azure Document Intelligence conversion: {e}")
|
|
96
|
+
return None
|
|
97
|
+
|
|
98
|
+
async def recognize_azure_speech_to_text_from_file(self, file_path: str) -> Optional[str]:
|
|
99
|
+
"""
|
|
100
|
+
Recognize speech from an audio file with automatic language detection
|
|
101
|
+
across the top 6 spoken languages globally.
|
|
102
|
+
|
|
103
|
+
Args:
|
|
104
|
+
file_path (str): Path to the audio file.
|
|
105
|
+
|
|
106
|
+
Returns:
|
|
107
|
+
Optional[str]: Transcribed text if successful, None otherwise.
|
|
108
|
+
"""
|
|
109
|
+
if not self.azure_speech_config:
|
|
110
|
+
logging.info("Azure Speech Service not configured, skipping speech recognition.")
|
|
111
|
+
return None
|
|
112
|
+
|
|
113
|
+
try:
|
|
114
|
+
audio_config = speechsdk.AudioConfig(filename=file_path)
|
|
115
|
+
languages = ["en-US", "zh-CN", "hi-IN", "es-ES"]
|
|
116
|
+
|
|
117
|
+
# Configure auto language detection with the specified languages
|
|
118
|
+
auto_detect_source_language_config = (
|
|
119
|
+
speechsdk.languageconfig.AutoDetectSourceLanguageConfig(languages=languages)
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
# Create a speech recognizer with the auto language detection configuration
|
|
123
|
+
speech_recognizer = speechsdk.SpeechRecognizer(
|
|
124
|
+
speech_config=self.azure_speech_config,
|
|
125
|
+
audio_config=audio_config,
|
|
126
|
+
auto_detect_source_language_config=auto_detect_source_language_config,
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
# Perform speech recognition
|
|
130
|
+
result = await speech_recognizer.recognize_once_async()
|
|
131
|
+
|
|
132
|
+
# Check the result
|
|
133
|
+
if result.reason == speechsdk.ResultReason.RecognizedSpeech:
|
|
134
|
+
# Retrieve the detected language
|
|
135
|
+
detected_language = result.properties.get(
|
|
136
|
+
speechsdk.PropertyId.SpeechServiceConnection_AutoDetectSourceLanguageResult,
|
|
137
|
+
"Unknown",
|
|
138
|
+
)
|
|
139
|
+
logging.debug(f"Detected Language {detected_language}")
|
|
140
|
+
return result.text
|
|
141
|
+
|
|
142
|
+
elif result.reason == speechsdk.ResultReason.NoMatch:
|
|
143
|
+
logging.warning("No speech could be recognized from the audio.")
|
|
144
|
+
return None
|
|
145
|
+
|
|
146
|
+
elif result.reason == speechsdk.ResultReason.Canceled:
|
|
147
|
+
cancellation_details = speechsdk.CancellationDetails(result)
|
|
148
|
+
logging.error(
|
|
149
|
+
f"Speech Recognition canceled: {cancellation_details.reason}. "
|
|
150
|
+
f"Error details: {cancellation_details.error_details}"
|
|
151
|
+
)
|
|
152
|
+
return None
|
|
153
|
+
|
|
154
|
+
else:
|
|
155
|
+
logging.error("Unknown error occurred during speech recognition.")
|
|
156
|
+
return None
|
|
157
|
+
|
|
158
|
+
except Exception as e:
|
|
159
|
+
logging.error(f"An error occurred during speech recognition: {e}")
|
|
160
|
+
return None
|
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
import base64
|
|
3
|
+
|
|
4
|
+
# import concurrent.futures
|
|
5
|
+
import mimetypes
|
|
6
|
+
import os
|
|
7
|
+
import re
|
|
8
|
+
import tempfile
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Optional, Tuple
|
|
11
|
+
|
|
12
|
+
import fitz
|
|
13
|
+
from langchain_core.messages import HumanMessage
|
|
14
|
+
from langchain_openai import AzureChatOpenAI
|
|
15
|
+
from PIL import Image as PILImage
|
|
16
|
+
|
|
17
|
+
from ..common.logger import logger
|
|
18
|
+
from ..common.utils import clean_markdown, ensure_minimum_content
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class GPT4oMiniVision:
|
|
22
|
+
def __init__(self=None):
|
|
23
|
+
azure_key = os.getenv("AZURE_OPENAI_API_KEY")
|
|
24
|
+
api_version = os.getenv("AZURE_OPENAI_API_VERSION")
|
|
25
|
+
|
|
26
|
+
GPT4OMINI_DEPLOYMENT_NAME: str = os.getenv("GPT4oMINI_DEPLOYMENT_NAME", "")
|
|
27
|
+
COMPLETION_TOKENS: int = 4096
|
|
28
|
+
|
|
29
|
+
self.client = None
|
|
30
|
+
if GPT4OMINI_DEPLOYMENT_NAME:
|
|
31
|
+
try:
|
|
32
|
+
from pydantic import BaseModel
|
|
33
|
+
|
|
34
|
+
class ImageSchema(BaseModel):
|
|
35
|
+
ocr: str
|
|
36
|
+
analysis: str
|
|
37
|
+
|
|
38
|
+
llm = AzureChatOpenAI(
|
|
39
|
+
deployment_name=GPT4OMINI_DEPLOYMENT_NAME,
|
|
40
|
+
temperature=0,
|
|
41
|
+
max_tokens=COMPLETION_TOKENS,
|
|
42
|
+
streaming=False,
|
|
43
|
+
api_version=api_version,
|
|
44
|
+
api_key=azure_key,
|
|
45
|
+
cache=False,
|
|
46
|
+
)
|
|
47
|
+
self.client = llm.with_structured_output(ImageSchema)
|
|
48
|
+
logger.info("GPT-4o-mini client initialized successfully.")
|
|
49
|
+
except Exception as e:
|
|
50
|
+
logger.warning(f"Failed to initialize GPT-4o-mini client: {e}")
|
|
51
|
+
self.client = None
|
|
52
|
+
|
|
53
|
+
def _build_image_url_block(self, file_or_url: str) -> dict:
|
|
54
|
+
if isinstance(file_or_url, Path):
|
|
55
|
+
file_or_url = str(file_or_url)
|
|
56
|
+
|
|
57
|
+
# If the input is a URL, return it as an image_url block
|
|
58
|
+
if re.match(r"^https?://", file_or_url, re.IGNORECASE):
|
|
59
|
+
return {
|
|
60
|
+
"type": "image_url",
|
|
61
|
+
"image_url": {"url": file_or_url},
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
path = Path(file_or_url)
|
|
65
|
+
if not path.is_file():
|
|
66
|
+
raise ValueError(f"Local file not found: {file_or_url}")
|
|
67
|
+
|
|
68
|
+
try:
|
|
69
|
+
content_type, _ = mimetypes.guess_type(file_or_url)
|
|
70
|
+
if not content_type:
|
|
71
|
+
content_type = "image/jpeg"
|
|
72
|
+
logger.warning(
|
|
73
|
+
f"Could not determine content type for {file_or_url}, defaulting to image/jpeg"
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
# Convert the image to JPEG and encode it in Base64
|
|
77
|
+
img = PILImage.open(file_or_url)
|
|
78
|
+
with tempfile.NamedTemporaryFile(suffix=".jpg", delete=False) as tmp_file:
|
|
79
|
+
img.convert("RGB").save(tmp_file.name, "JPEG")
|
|
80
|
+
jpeg_path = tmp_file.name
|
|
81
|
+
|
|
82
|
+
with open(jpeg_path, "rb") as f:
|
|
83
|
+
b64_data = base64.b64encode(f.read()).decode("utf-8")
|
|
84
|
+
|
|
85
|
+
os.unlink(jpeg_path) # Delete the temporary JPEG file
|
|
86
|
+
|
|
87
|
+
return {
|
|
88
|
+
"type": "image_url",
|
|
89
|
+
"image_url": {"url": f"data:{content_type};base64,{b64_data}"},
|
|
90
|
+
}
|
|
91
|
+
except Exception as e:
|
|
92
|
+
logger.error(f"Error building image block for {file_or_url}: {e}")
|
|
93
|
+
raise
|
|
94
|
+
|
|
95
|
+
async def process_image(self, file_or_url: str, prompt: Optional[str] = None) -> Optional[str]:
|
|
96
|
+
if not prompt:
|
|
97
|
+
prompt = (
|
|
98
|
+
"Perform accurate OCR: answer with the markdown of the content of the image. Visually appealing markdown, nothing else. "
|
|
99
|
+
"Perform Image Analysis: answer with a two line image analysis explaining what you see."
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
try:
|
|
103
|
+
image_block = self._build_image_url_block(file_or_url)
|
|
104
|
+
message_content = [
|
|
105
|
+
{"type": "text", "text": prompt},
|
|
106
|
+
image_block,
|
|
107
|
+
]
|
|
108
|
+
message = HumanMessage(content=message_content)
|
|
109
|
+
response = self.client.invoke([message])
|
|
110
|
+
md_content = clean_markdown(response.analysis)
|
|
111
|
+
md_content += "\n\n" + clean_markdown(response.ocr)
|
|
112
|
+
|
|
113
|
+
return md_content if ensure_minimum_content(md_content) else None
|
|
114
|
+
except Exception as e:
|
|
115
|
+
logger.error(f"process_image: Error during GPT-4o-mini image OCR: {e}")
|
|
116
|
+
return None
|
|
117
|
+
|
|
118
|
+
async def process_scanned_pdf_concurrent(self, file_path: str) -> Optional[str]:
|
|
119
|
+
if not self.client:
|
|
120
|
+
return None
|
|
121
|
+
|
|
122
|
+
try:
|
|
123
|
+
doc = fitz.open(file_path)
|
|
124
|
+
num_pages = doc.page_count
|
|
125
|
+
results = []
|
|
126
|
+
|
|
127
|
+
async def ocr_page(page_index: int) -> Tuple[int, str]:
|
|
128
|
+
try:
|
|
129
|
+
page = doc.load_page(page_index)
|
|
130
|
+
pix = page.get_pixmap(dpi=150)
|
|
131
|
+
|
|
132
|
+
fd, png_path = tempfile.mkstemp(prefix=f"{page_index:3}-", suffix=".png")
|
|
133
|
+
os.close(fd)
|
|
134
|
+
pix.save(png_path)
|
|
135
|
+
|
|
136
|
+
try:
|
|
137
|
+
file_name = os.path.basename(file_path)
|
|
138
|
+
partial_md = await self.process_image(png_path)
|
|
139
|
+
if partial_md:
|
|
140
|
+
logger.debug(
|
|
141
|
+
f"ocr_page: {file_name} - page {page_index}: returned {len(partial_md)} characters"
|
|
142
|
+
)
|
|
143
|
+
else:
|
|
144
|
+
logger.warning(
|
|
145
|
+
f"ocr_page: {file_name} - page {page_index}: no OCR result"
|
|
146
|
+
)
|
|
147
|
+
partial_md = "[No OCR result]"
|
|
148
|
+
return (page_index, f"## Page {page_index + 1}\n\n{partial_md}")
|
|
149
|
+
finally:
|
|
150
|
+
try:
|
|
151
|
+
if os.path.exists(png_path):
|
|
152
|
+
# Remove the temporary PNG file after processing
|
|
153
|
+
logger.debug(f"Removing temporary PNG file: {png_path}")
|
|
154
|
+
Path(png_path).unlink(missing_ok=True)
|
|
155
|
+
except Exception as e:
|
|
156
|
+
logger.error(f"ocr_page: Error removing temporary PNG file: {e}")
|
|
157
|
+
except Exception as e:
|
|
158
|
+
logger.error(f"ocr_page: Error processing page {page_index}: {e}")
|
|
159
|
+
return (page_index, None)
|
|
160
|
+
|
|
161
|
+
logger.info(
|
|
162
|
+
f"process_scanned_pdf_concurrent: Starting concurrent GPT-4o-mini OCR on PDF: '{file_path}'"
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
# run tasks concurrently
|
|
166
|
+
tasks = [ocr_page(i) for i in range(num_pages)]
|
|
167
|
+
results = await asyncio.gather(*tasks)
|
|
168
|
+
|
|
169
|
+
combined_md = clean_markdown("\n\n".join(block for _, block in results))
|
|
170
|
+
return combined_md if ensure_minimum_content(combined_md) else None
|
|
171
|
+
|
|
172
|
+
except Exception as e:
|
|
173
|
+
logger.error(f"Concurrent GPT-4o-mini OCR error for PDF {file_path}: {e}")
|
|
174
|
+
return None
|
|
175
|
+
|
|
176
|
+
async def process_scanned_pdf_simple(self, file_path: str) -> Optional[str]:
|
|
177
|
+
if not self.client:
|
|
178
|
+
return None
|
|
179
|
+
|
|
180
|
+
try:
|
|
181
|
+
doc = fitz.open(file_path)
|
|
182
|
+
all_md_blocks = []
|
|
183
|
+
|
|
184
|
+
logger.info(f"Starting simple GPT-4o-mini OCR on PDF: '{file_path}'")
|
|
185
|
+
for i in range(doc.page_count):
|
|
186
|
+
page = doc.load_page(i)
|
|
187
|
+
pix = page.get_pixmap(dpi=150)
|
|
188
|
+
with tempfile.NamedTemporaryFile(suffix=".png", delete=False) as tmp_file:
|
|
189
|
+
png_path = tmp_file.name
|
|
190
|
+
pix.save(png_path)
|
|
191
|
+
try:
|
|
192
|
+
partial_md = await self.process_image(png_path)
|
|
193
|
+
if partial_md:
|
|
194
|
+
block = f"## Page {i + 1}\n\n{partial_md}"
|
|
195
|
+
else:
|
|
196
|
+
block = f"## Page {i + 1}\n\n[No OCR result]"
|
|
197
|
+
all_md_blocks.append(block)
|
|
198
|
+
finally:
|
|
199
|
+
try:
|
|
200
|
+
Path(png_path).unlink(missing_ok=True)
|
|
201
|
+
except Exception as e:
|
|
202
|
+
logger.error(f"Error removing temporary PNG file in simple OCR: {e}")
|
|
203
|
+
|
|
204
|
+
combined_md = clean_markdown("\n\n".join(all_md_blocks))
|
|
205
|
+
return combined_md if ensure_minimum_content(combined_md) else None
|
|
206
|
+
|
|
207
|
+
except Exception as e:
|
|
208
|
+
logger.error(f"Simple GPT-4o-mini OCR error for PDF {file_path}: {e}")
|
|
209
|
+
return None
|