markitdown-ocr 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- markitdown_ocr/__about__.py +4 -0
- markitdown_ocr/__init__.py +31 -0
- markitdown_ocr/_docx_converter_with_ocr.py +189 -0
- markitdown_ocr/_ocr_service.py +110 -0
- markitdown_ocr/_pdf_converter_with_ocr.py +422 -0
- markitdown_ocr/_plugin.py +68 -0
- markitdown_ocr/_pptx_converter_with_ocr.py +249 -0
- markitdown_ocr/_xlsx_converter_with_ocr.py +225 -0
- markitdown_ocr-0.1.0.dist-info/METADATA +233 -0
- markitdown_ocr-0.1.0.dist-info/RECORD +13 -0
- markitdown_ocr-0.1.0.dist-info/WHEEL +4 -0
- markitdown_ocr-0.1.0.dist-info/entry_points.txt +2 -0
- markitdown_ocr-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,249 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Enhanced PPTX Converter with improved OCR support.
|
|
3
|
+
Already has LLM-based image description, this enhances it with traditional OCR fallback.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import io
|
|
7
|
+
import sys
|
|
8
|
+
from typing import Any, BinaryIO, Optional
|
|
9
|
+
|
|
10
|
+
from typing import BinaryIO, Any, Optional
|
|
11
|
+
|
|
12
|
+
from markitdown.converters import HtmlConverter
|
|
13
|
+
from markitdown import DocumentConverter, DocumentConverterResult, StreamInfo
|
|
14
|
+
from markitdown._exceptions import (
|
|
15
|
+
MissingDependencyException,
|
|
16
|
+
MISSING_DEPENDENCY_MESSAGE,
|
|
17
|
+
)
|
|
18
|
+
from ._ocr_service import LLMVisionOCRService
|
|
19
|
+
|
|
20
|
+
_dependency_exc_info = None
|
|
21
|
+
try:
|
|
22
|
+
import pptx
|
|
23
|
+
except ImportError:
|
|
24
|
+
_dependency_exc_info = sys.exc_info()
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class PptxConverterWithOCR(DocumentConverter):
|
|
28
|
+
"""Enhanced PPTX Converter with OCR fallback."""
|
|
29
|
+
|
|
30
|
+
def __init__(self, ocr_service: Optional[LLMVisionOCRService] = None):
|
|
31
|
+
super().__init__()
|
|
32
|
+
self._html_converter = HtmlConverter()
|
|
33
|
+
self.ocr_service = ocr_service
|
|
34
|
+
|
|
35
|
+
def accepts(
|
|
36
|
+
self,
|
|
37
|
+
file_stream: BinaryIO,
|
|
38
|
+
stream_info: StreamInfo,
|
|
39
|
+
**kwargs: Any,
|
|
40
|
+
) -> bool:
|
|
41
|
+
mimetype = (stream_info.mimetype or "").lower()
|
|
42
|
+
extension = (stream_info.extension or "").lower()
|
|
43
|
+
|
|
44
|
+
if extension == ".pptx":
|
|
45
|
+
return True
|
|
46
|
+
|
|
47
|
+
if mimetype.startswith(
|
|
48
|
+
"application/vnd.openxmlformats-officedocument.presentationml"
|
|
49
|
+
):
|
|
50
|
+
return True
|
|
51
|
+
|
|
52
|
+
return False
|
|
53
|
+
|
|
54
|
+
def convert(
|
|
55
|
+
self,
|
|
56
|
+
file_stream: BinaryIO,
|
|
57
|
+
stream_info: StreamInfo,
|
|
58
|
+
**kwargs: Any,
|
|
59
|
+
) -> DocumentConverterResult:
|
|
60
|
+
if _dependency_exc_info is not None:
|
|
61
|
+
raise MissingDependencyException(
|
|
62
|
+
MISSING_DEPENDENCY_MESSAGE.format(
|
|
63
|
+
converter=type(self).__name__,
|
|
64
|
+
extension=".pptx",
|
|
65
|
+
feature="pptx",
|
|
66
|
+
)
|
|
67
|
+
) from _dependency_exc_info[1].with_traceback(
|
|
68
|
+
_dependency_exc_info[2]
|
|
69
|
+
) # type: ignore[union-attr]
|
|
70
|
+
|
|
71
|
+
# Get OCR service (from kwargs or instance)
|
|
72
|
+
ocr_service: Optional[LLMVisionOCRService] = (
|
|
73
|
+
kwargs.get("ocr_service") or self.ocr_service
|
|
74
|
+
)
|
|
75
|
+
llm_client = kwargs.get("llm_client")
|
|
76
|
+
|
|
77
|
+
presentation = pptx.Presentation(file_stream)
|
|
78
|
+
md_content = ""
|
|
79
|
+
slide_num = 0
|
|
80
|
+
|
|
81
|
+
for slide in presentation.slides:
|
|
82
|
+
slide_num += 1
|
|
83
|
+
md_content += f"\\n\\n<!-- Slide number: {slide_num} -->\\n"
|
|
84
|
+
|
|
85
|
+
title = slide.shapes.title
|
|
86
|
+
|
|
87
|
+
def get_shape_content(shape, **kwargs):
|
|
88
|
+
nonlocal md_content
|
|
89
|
+
|
|
90
|
+
# Pictures
|
|
91
|
+
if self._is_picture(shape):
|
|
92
|
+
# Get image data
|
|
93
|
+
image_stream = io.BytesIO(shape.image.blob)
|
|
94
|
+
|
|
95
|
+
# Try LLM description first if available
|
|
96
|
+
llm_description = ""
|
|
97
|
+
if llm_client and kwargs.get("llm_model"):
|
|
98
|
+
try:
|
|
99
|
+
from ._llm_caption import llm_caption
|
|
100
|
+
|
|
101
|
+
image_filename = shape.image.filename
|
|
102
|
+
image_extension = None
|
|
103
|
+
if image_filename:
|
|
104
|
+
import os
|
|
105
|
+
|
|
106
|
+
image_extension = os.path.splitext(image_filename)[1]
|
|
107
|
+
|
|
108
|
+
image_stream_info = StreamInfo(
|
|
109
|
+
mimetype=shape.image.content_type,
|
|
110
|
+
extension=image_extension,
|
|
111
|
+
filename=image_filename,
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
llm_description = llm_caption(
|
|
115
|
+
image_stream,
|
|
116
|
+
image_stream_info,
|
|
117
|
+
client=llm_client,
|
|
118
|
+
model=kwargs.get("llm_model"),
|
|
119
|
+
prompt=kwargs.get("llm_prompt"),
|
|
120
|
+
)
|
|
121
|
+
except Exception:
|
|
122
|
+
pass
|
|
123
|
+
|
|
124
|
+
# Try OCR if LLM failed or not available
|
|
125
|
+
ocr_text = ""
|
|
126
|
+
if not llm_description and ocr_service:
|
|
127
|
+
try:
|
|
128
|
+
image_stream.seek(0)
|
|
129
|
+
ocr_result = ocr_service.extract_text(image_stream)
|
|
130
|
+
if ocr_result.text.strip():
|
|
131
|
+
ocr_text = ocr_result.text.strip()
|
|
132
|
+
except Exception:
|
|
133
|
+
pass
|
|
134
|
+
|
|
135
|
+
# Format extracted content using unified OCR block format
|
|
136
|
+
content = (llm_description or ocr_text or "").strip()
|
|
137
|
+
if content:
|
|
138
|
+
md_content += f"\n*[Image OCR]\n{content}\n[End OCR]*\n"
|
|
139
|
+
|
|
140
|
+
# Tables
|
|
141
|
+
if self._is_table(shape):
|
|
142
|
+
md_content += self._convert_table_to_markdown(shape.table, **kwargs)
|
|
143
|
+
|
|
144
|
+
# Charts
|
|
145
|
+
if shape.has_chart:
|
|
146
|
+
md_content += self._convert_chart_to_markdown(shape.chart)
|
|
147
|
+
|
|
148
|
+
# Text areas
|
|
149
|
+
elif shape.has_text_frame:
|
|
150
|
+
if shape == title:
|
|
151
|
+
md_content += "# " + shape.text.lstrip() + "\\n"
|
|
152
|
+
else:
|
|
153
|
+
md_content += shape.text + "\\n"
|
|
154
|
+
|
|
155
|
+
# Group Shapes
|
|
156
|
+
if shape.shape_type == pptx.enum.shapes.MSO_SHAPE_TYPE.GROUP:
|
|
157
|
+
sorted_shapes = sorted(
|
|
158
|
+
shape.shapes,
|
|
159
|
+
key=lambda x: (
|
|
160
|
+
float("-inf") if not x.top else x.top,
|
|
161
|
+
float("-inf") if not x.left else x.left,
|
|
162
|
+
),
|
|
163
|
+
)
|
|
164
|
+
for subshape in sorted_shapes:
|
|
165
|
+
get_shape_content(subshape, **kwargs)
|
|
166
|
+
|
|
167
|
+
sorted_shapes = sorted(
|
|
168
|
+
slide.shapes,
|
|
169
|
+
key=lambda x: (
|
|
170
|
+
float("-inf") if not x.top else x.top,
|
|
171
|
+
float("-inf") if not x.left else x.left,
|
|
172
|
+
),
|
|
173
|
+
)
|
|
174
|
+
for shape in sorted_shapes:
|
|
175
|
+
get_shape_content(shape, **kwargs)
|
|
176
|
+
|
|
177
|
+
md_content = md_content.strip()
|
|
178
|
+
|
|
179
|
+
if slide.has_notes_slide:
|
|
180
|
+
md_content += "\\n\\n### Notes:\\n"
|
|
181
|
+
notes_frame = slide.notes_slide.notes_text_frame
|
|
182
|
+
if notes_frame is not None:
|
|
183
|
+
md_content += notes_frame.text
|
|
184
|
+
md_content = md_content.strip()
|
|
185
|
+
|
|
186
|
+
return DocumentConverterResult(markdown=md_content.strip())
|
|
187
|
+
|
|
188
|
+
def _is_picture(self, shape):
|
|
189
|
+
if shape.shape_type == pptx.enum.shapes.MSO_SHAPE_TYPE.PICTURE:
|
|
190
|
+
return True
|
|
191
|
+
if shape.shape_type == pptx.enum.shapes.MSO_SHAPE_TYPE.PLACEHOLDER:
|
|
192
|
+
if hasattr(shape, "image"):
|
|
193
|
+
return True
|
|
194
|
+
return False
|
|
195
|
+
|
|
196
|
+
def _is_table(self, shape):
|
|
197
|
+
if shape.shape_type == pptx.enum.shapes.MSO_SHAPE_TYPE.TABLE:
|
|
198
|
+
return True
|
|
199
|
+
return False
|
|
200
|
+
|
|
201
|
+
def _convert_table_to_markdown(self, table, **kwargs):
|
|
202
|
+
import html
|
|
203
|
+
|
|
204
|
+
html_table = "<html><body><table>"
|
|
205
|
+
first_row = True
|
|
206
|
+
for row in table.rows:
|
|
207
|
+
html_table += "<tr>"
|
|
208
|
+
for cell in row.cells:
|
|
209
|
+
if first_row:
|
|
210
|
+
html_table += "<th>" + html.escape(cell.text) + "</th>"
|
|
211
|
+
else:
|
|
212
|
+
html_table += "<td>" + html.escape(cell.text) + "</td>"
|
|
213
|
+
html_table += "</tr>"
|
|
214
|
+
first_row = False
|
|
215
|
+
html_table += "</table></body></html>"
|
|
216
|
+
|
|
217
|
+
return (
|
|
218
|
+
self._html_converter.convert_string(html_table, **kwargs).markdown.strip()
|
|
219
|
+
+ "\\n"
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
def _convert_chart_to_markdown(self, chart):
|
|
223
|
+
try:
|
|
224
|
+
md = "\\n\\n### Chart"
|
|
225
|
+
if chart.has_title:
|
|
226
|
+
md += f": {chart.chart_title.text_frame.text}"
|
|
227
|
+
md += "\\n\\n"
|
|
228
|
+
data = []
|
|
229
|
+
category_names = [c.label for c in chart.plots[0].categories]
|
|
230
|
+
series_names = [s.name for s in chart.series]
|
|
231
|
+
data.append(["Category"] + series_names)
|
|
232
|
+
|
|
233
|
+
for idx, category in enumerate(category_names):
|
|
234
|
+
row = [category]
|
|
235
|
+
for series in chart.series:
|
|
236
|
+
row.append(series.values[idx])
|
|
237
|
+
data.append(row)
|
|
238
|
+
|
|
239
|
+
markdown_table = []
|
|
240
|
+
for row in data:
|
|
241
|
+
markdown_table.append("| " + " | ".join(map(str, row)) + " |")
|
|
242
|
+
header = markdown_table[0]
|
|
243
|
+
separator = "|" + "|".join(["---"] * len(data[0])) + "|"
|
|
244
|
+
return md + "\\n".join([header, separator] + markdown_table[1:])
|
|
245
|
+
except ValueError as e:
|
|
246
|
+
if "unsupported plot type" in str(e):
|
|
247
|
+
return "\\n\\n[unsupported chart]\\n\\n"
|
|
248
|
+
except Exception:
|
|
249
|
+
return "\\n\\n[unsupported chart]\\n\\n"
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Enhanced XLSX Converter with OCR support for embedded images.
|
|
3
|
+
Extracts images from Excel spreadsheets and performs OCR while maintaining cell context.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import io
|
|
7
|
+
import sys
|
|
8
|
+
from typing import Any, BinaryIO, Optional
|
|
9
|
+
|
|
10
|
+
from markitdown.converters import HtmlConverter
|
|
11
|
+
from markitdown import DocumentConverter, DocumentConverterResult, StreamInfo
|
|
12
|
+
from markitdown._exceptions import (
|
|
13
|
+
MissingDependencyException,
|
|
14
|
+
MISSING_DEPENDENCY_MESSAGE,
|
|
15
|
+
)
|
|
16
|
+
from ._ocr_service import LLMVisionOCRService
|
|
17
|
+
|
|
18
|
+
# Try loading dependencies
|
|
19
|
+
_xlsx_dependency_exc_info = None
|
|
20
|
+
try:
|
|
21
|
+
import pandas as pd
|
|
22
|
+
from openpyxl import load_workbook
|
|
23
|
+
except ImportError:
|
|
24
|
+
_xlsx_dependency_exc_info = sys.exc_info()
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class XlsxConverterWithOCR(DocumentConverter):
|
|
28
|
+
"""
|
|
29
|
+
Enhanced XLSX Converter with OCR support for embedded images.
|
|
30
|
+
Extracts images with their cell positions and performs OCR.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
def __init__(self, ocr_service: Optional[LLMVisionOCRService] = None):
|
|
34
|
+
super().__init__()
|
|
35
|
+
self._html_converter = HtmlConverter()
|
|
36
|
+
self.ocr_service = ocr_service
|
|
37
|
+
|
|
38
|
+
def accepts(
|
|
39
|
+
self,
|
|
40
|
+
file_stream: BinaryIO,
|
|
41
|
+
stream_info: StreamInfo,
|
|
42
|
+
**kwargs: Any,
|
|
43
|
+
) -> bool:
|
|
44
|
+
mimetype = (stream_info.mimetype or "").lower()
|
|
45
|
+
extension = (stream_info.extension or "").lower()
|
|
46
|
+
|
|
47
|
+
if extension == ".xlsx":
|
|
48
|
+
return True
|
|
49
|
+
|
|
50
|
+
if mimetype.startswith(
|
|
51
|
+
"application/vnd.openxmlformats-officedocument.spreadsheetml"
|
|
52
|
+
):
|
|
53
|
+
return True
|
|
54
|
+
|
|
55
|
+
return False
|
|
56
|
+
|
|
57
|
+
def convert(
|
|
58
|
+
self,
|
|
59
|
+
file_stream: BinaryIO,
|
|
60
|
+
stream_info: StreamInfo,
|
|
61
|
+
**kwargs: Any,
|
|
62
|
+
) -> DocumentConverterResult:
|
|
63
|
+
if _xlsx_dependency_exc_info is not None:
|
|
64
|
+
raise MissingDependencyException(
|
|
65
|
+
MISSING_DEPENDENCY_MESSAGE.format(
|
|
66
|
+
converter=type(self).__name__,
|
|
67
|
+
extension=".xlsx",
|
|
68
|
+
feature="xlsx",
|
|
69
|
+
)
|
|
70
|
+
) from _xlsx_dependency_exc_info[1].with_traceback(
|
|
71
|
+
_xlsx_dependency_exc_info[2]
|
|
72
|
+
) # type: ignore[union-attr]
|
|
73
|
+
|
|
74
|
+
# Get OCR service if available (from kwargs or instance)
|
|
75
|
+
ocr_service: Optional[LLMVisionOCRService] = (
|
|
76
|
+
kwargs.get("ocr_service") or self.ocr_service
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
if ocr_service:
|
|
80
|
+
# Remove ocr_service from kwargs to avoid duplicate argument error
|
|
81
|
+
kwargs_without_ocr = {k: v for k, v in kwargs.items() if k != "ocr_service"}
|
|
82
|
+
return self._convert_with_ocr(
|
|
83
|
+
file_stream, ocr_service, **kwargs_without_ocr
|
|
84
|
+
)
|
|
85
|
+
else:
|
|
86
|
+
return self._convert_standard(file_stream, **kwargs)
|
|
87
|
+
|
|
88
|
+
def _convert_standard(
|
|
89
|
+
self, file_stream: BinaryIO, **kwargs: Any
|
|
90
|
+
) -> DocumentConverterResult:
|
|
91
|
+
"""Standard conversion without OCR."""
|
|
92
|
+
file_stream.seek(0)
|
|
93
|
+
sheets = pd.read_excel(file_stream, sheet_name=None, engine="openpyxl")
|
|
94
|
+
md_content = ""
|
|
95
|
+
|
|
96
|
+
for sheet_name in sheets:
|
|
97
|
+
md_content += f"## {sheet_name}\n"
|
|
98
|
+
html_content = sheets[sheet_name].to_html(index=False)
|
|
99
|
+
md_content += (
|
|
100
|
+
self._html_converter.convert_string(
|
|
101
|
+
html_content, **kwargs
|
|
102
|
+
).markdown.strip()
|
|
103
|
+
+ "\n\n"
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
return DocumentConverterResult(markdown=md_content.strip())
|
|
107
|
+
|
|
108
|
+
def _convert_with_ocr(
|
|
109
|
+
self, file_stream: BinaryIO, ocr_service: LLMVisionOCRService, **kwargs: Any
|
|
110
|
+
) -> DocumentConverterResult:
|
|
111
|
+
"""Convert XLSX with image OCR."""
|
|
112
|
+
file_stream.seek(0)
|
|
113
|
+
wb = load_workbook(file_stream)
|
|
114
|
+
|
|
115
|
+
md_content = ""
|
|
116
|
+
|
|
117
|
+
for sheet_name in wb.sheetnames:
|
|
118
|
+
sheet = wb[sheet_name]
|
|
119
|
+
md_content += f"## {sheet_name}\n\n"
|
|
120
|
+
|
|
121
|
+
# Convert sheet data to markdown table
|
|
122
|
+
file_stream.seek(0)
|
|
123
|
+
try:
|
|
124
|
+
df = pd.read_excel(
|
|
125
|
+
file_stream, sheet_name=sheet_name, engine="openpyxl"
|
|
126
|
+
)
|
|
127
|
+
html_content = df.to_html(index=False)
|
|
128
|
+
md_content += (
|
|
129
|
+
self._html_converter.convert_string(
|
|
130
|
+
html_content, **kwargs
|
|
131
|
+
).markdown.strip()
|
|
132
|
+
+ "\n\n"
|
|
133
|
+
)
|
|
134
|
+
except Exception:
|
|
135
|
+
# If pandas fails, just skip the table
|
|
136
|
+
pass
|
|
137
|
+
|
|
138
|
+
# Extract and OCR images in this sheet
|
|
139
|
+
images_with_ocr = self._extract_and_ocr_sheet_images(sheet, ocr_service)
|
|
140
|
+
|
|
141
|
+
if images_with_ocr:
|
|
142
|
+
md_content += "### Images in this sheet:\n\n"
|
|
143
|
+
for img_info in images_with_ocr:
|
|
144
|
+
ocr_text = img_info["ocr_text"]
|
|
145
|
+
md_content += f"*[Image OCR]\n{ocr_text}\n[End OCR]*\n\n"
|
|
146
|
+
|
|
147
|
+
return DocumentConverterResult(markdown=md_content.strip())
|
|
148
|
+
|
|
149
|
+
def _extract_and_ocr_sheet_images(
|
|
150
|
+
self, sheet: Any, ocr_service: LLMVisionOCRService
|
|
151
|
+
) -> list[dict]:
|
|
152
|
+
"""
|
|
153
|
+
Extract and OCR images from an Excel sheet.
|
|
154
|
+
|
|
155
|
+
Args:
|
|
156
|
+
sheet: openpyxl worksheet
|
|
157
|
+
ocr_service: OCR service
|
|
158
|
+
|
|
159
|
+
Returns:
|
|
160
|
+
List of dicts with 'cell_ref' and 'ocr_text'
|
|
161
|
+
"""
|
|
162
|
+
results = []
|
|
163
|
+
|
|
164
|
+
try:
|
|
165
|
+
# Check if sheet has images
|
|
166
|
+
if hasattr(sheet, "_images"):
|
|
167
|
+
for img in sheet._images:
|
|
168
|
+
try:
|
|
169
|
+
# Get image data
|
|
170
|
+
if hasattr(img, "_data"):
|
|
171
|
+
image_data = img._data()
|
|
172
|
+
elif hasattr(img, "image"):
|
|
173
|
+
# Some versions store it differently
|
|
174
|
+
image_data = img.image
|
|
175
|
+
else:
|
|
176
|
+
continue
|
|
177
|
+
|
|
178
|
+
# Create image stream
|
|
179
|
+
image_stream = io.BytesIO(image_data)
|
|
180
|
+
|
|
181
|
+
# Get cell reference
|
|
182
|
+
cell_ref = "unknown"
|
|
183
|
+
if hasattr(img, "anchor"):
|
|
184
|
+
anchor = img.anchor
|
|
185
|
+
if hasattr(anchor, "_from"):
|
|
186
|
+
from_cell = anchor._from
|
|
187
|
+
if hasattr(from_cell, "col") and hasattr(
|
|
188
|
+
from_cell, "row"
|
|
189
|
+
):
|
|
190
|
+
# Convert column number to letter
|
|
191
|
+
col_letter = self._column_number_to_letter(
|
|
192
|
+
from_cell.col
|
|
193
|
+
)
|
|
194
|
+
cell_ref = f"{col_letter}{from_cell.row + 1}"
|
|
195
|
+
|
|
196
|
+
# Perform OCR
|
|
197
|
+
ocr_result = ocr_service.extract_text(image_stream)
|
|
198
|
+
|
|
199
|
+
if ocr_result.text.strip():
|
|
200
|
+
results.append(
|
|
201
|
+
{
|
|
202
|
+
"cell_ref": cell_ref,
|
|
203
|
+
"ocr_text": ocr_result.text.strip(),
|
|
204
|
+
"backend": ocr_result.backend_used,
|
|
205
|
+
}
|
|
206
|
+
)
|
|
207
|
+
|
|
208
|
+
except Exception:
|
|
209
|
+
continue
|
|
210
|
+
|
|
211
|
+
except Exception:
|
|
212
|
+
pass
|
|
213
|
+
|
|
214
|
+
return results
|
|
215
|
+
|
|
216
|
+
@staticmethod
|
|
217
|
+
def _column_number_to_letter(n: int) -> str:
|
|
218
|
+
"""Convert column number to Excel column letter (0-indexed)."""
|
|
219
|
+
result = ""
|
|
220
|
+
n = n + 1 # Make 1-indexed
|
|
221
|
+
while n > 0:
|
|
222
|
+
n -= 1
|
|
223
|
+
result = chr(65 + (n % 26)) + result
|
|
224
|
+
n //= 26
|
|
225
|
+
return result
|
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: markitdown-ocr
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: OCR plugin for MarkItDown - Extracts text from images in PDF, DOCX, PPTX, and XLSX via LLM Vision
|
|
5
|
+
Project-URL: Documentation, https://github.com/microsoft/markitdown#readme
|
|
6
|
+
Project-URL: Issues, https://github.com/microsoft/markitdown/issues
|
|
7
|
+
Project-URL: Source, https://github.com/microsoft/markitdown
|
|
8
|
+
Author-email: Contributors <noreply@github.com>
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: docx,llm,markitdown,ocr,pdf,pptx,vision,xlsx
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Programming Language :: Python
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Programming Language :: Python :: Implementation :: CPython
|
|
19
|
+
Requires-Python: >=3.10
|
|
20
|
+
Requires-Dist: mammoth~=1.11.0
|
|
21
|
+
Requires-Dist: markitdown>=0.1.0
|
|
22
|
+
Requires-Dist: openpyxl
|
|
23
|
+
Requires-Dist: pandas
|
|
24
|
+
Requires-Dist: pdfminer-six>=20251230
|
|
25
|
+
Requires-Dist: pdfplumber>=0.11.9
|
|
26
|
+
Requires-Dist: pillow>=9.0.0
|
|
27
|
+
Requires-Dist: pymupdf>=1.24.0
|
|
28
|
+
Requires-Dist: python-docx
|
|
29
|
+
Requires-Dist: python-pptx
|
|
30
|
+
Provides-Extra: llm
|
|
31
|
+
Requires-Dist: openai>=1.0.0; extra == 'llm'
|
|
32
|
+
Description-Content-Type: text/markdown
|
|
33
|
+
|
|
34
|
+
# MarkItDown OCR Plugin
|
|
35
|
+
|
|
36
|
+
LLM Vision plugin for MarkItDown that extracts text from images embedded in PDF, DOCX, PPTX, and XLSX files.
|
|
37
|
+
|
|
38
|
+
Uses the same `llm_client` / `llm_model` pattern that MarkItDown already supports for image descriptions — no new ML libraries or binary dependencies required.
|
|
39
|
+
|
|
40
|
+
## Features
|
|
41
|
+
|
|
42
|
+
- **Enhanced PDF Converter**: Extracts text from images within PDFs, with full-page OCR fallback for scanned documents
|
|
43
|
+
- **Enhanced DOCX Converter**: OCR for images in Word documents
|
|
44
|
+
- **Enhanced PPTX Converter**: OCR for images in PowerPoint presentations
|
|
45
|
+
- **Enhanced XLSX Converter**: OCR for images in Excel spreadsheets
|
|
46
|
+
- **Context Preservation**: Maintains document structure and flow when inserting extracted text
|
|
47
|
+
|
|
48
|
+
## Installation
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
pip install markitdown-ocr
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
The plugin uses whatever OpenAI-compatible client you already have. Install one if you don't have it yet:
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
pip install openai
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
## Usage
|
|
61
|
+
|
|
62
|
+
### Command Line
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
markitdown document.pdf --use-plugins --llm-client openai --llm-model gpt-4o
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
### Python API
|
|
69
|
+
|
|
70
|
+
Pass `llm_client` and `llm_model` to `MarkItDown()` exactly as you would for image descriptions:
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
from markitdown import MarkItDown
|
|
74
|
+
from openai import OpenAI
|
|
75
|
+
|
|
76
|
+
md = MarkItDown(
|
|
77
|
+
enable_plugins=True,
|
|
78
|
+
llm_client=OpenAI(),
|
|
79
|
+
llm_model="gpt-4o",
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
result = md.convert("document_with_images.pdf")
|
|
83
|
+
print(result.text_content)
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
If no `llm_client` is provided the plugin still loads, but OCR is silently skipped — falling back to the standard built-in converter.
|
|
87
|
+
|
|
88
|
+
### Custom Prompt
|
|
89
|
+
|
|
90
|
+
Override the default extraction prompt for specialized documents:
|
|
91
|
+
|
|
92
|
+
```python
|
|
93
|
+
md = MarkItDown(
|
|
94
|
+
enable_plugins=True,
|
|
95
|
+
llm_client=OpenAI(),
|
|
96
|
+
llm_model="gpt-4o",
|
|
97
|
+
llm_prompt="Extract all text from this image, preserving table structure.",
|
|
98
|
+
)
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
### Any OpenAI-Compatible Client
|
|
102
|
+
|
|
103
|
+
Works with any client that follows the OpenAI API:
|
|
104
|
+
|
|
105
|
+
```python
|
|
106
|
+
from openai import AzureOpenAI
|
|
107
|
+
|
|
108
|
+
md = MarkItDown(
|
|
109
|
+
enable_plugins=True,
|
|
110
|
+
llm_client=AzureOpenAI(
|
|
111
|
+
api_key="...",
|
|
112
|
+
azure_endpoint="https://your-resource.openai.azure.com/",
|
|
113
|
+
api_version="2024-02-01",
|
|
114
|
+
),
|
|
115
|
+
llm_model="gpt-4o",
|
|
116
|
+
)
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
## How It Works
|
|
120
|
+
|
|
121
|
+
When `MarkItDown(enable_plugins=True, llm_client=..., llm_model=...)` is called:
|
|
122
|
+
|
|
123
|
+
1. MarkItDown discovers the plugin via the `markitdown.plugin` entry point group
|
|
124
|
+
2. It calls `register_converters()`, forwarding all kwargs including `llm_client` and `llm_model`
|
|
125
|
+
3. The plugin creates an `LLMVisionOCRService` from those kwargs
|
|
126
|
+
4. Four OCR-enhanced converters are registered at **priority -1.0** — before the built-in converters at priority 0.0
|
|
127
|
+
|
|
128
|
+
When a file is converted:
|
|
129
|
+
|
|
130
|
+
1. The OCR converter accepts the file
|
|
131
|
+
2. It extracts embedded images from the document
|
|
132
|
+
3. Each image is sent to the LLM with an extraction prompt
|
|
133
|
+
4. The returned text is inserted inline, preserving document structure
|
|
134
|
+
5. If the LLM call fails, conversion continues without that image's text
|
|
135
|
+
|
|
136
|
+
## Supported File Formats
|
|
137
|
+
|
|
138
|
+
### PDF
|
|
139
|
+
|
|
140
|
+
- Embedded images are extracted by position (via `page.images` / page XObjects) and OCR'd inline, interleaved with the surrounding text in vertical reading order.
|
|
141
|
+
- **Scanned PDFs** (pages with no extractable text) are detected automatically: each page is rendered at 300 DPI and sent to the LLM as a full-page image.
|
|
142
|
+
- **Malformed PDFs** that pdfplumber/pdfminer cannot open (e.g. truncated EOF) are retried with PyMuPDF page rendering, so content is still recovered.
|
|
143
|
+
|
|
144
|
+
### DOCX
|
|
145
|
+
|
|
146
|
+
- Images are extracted via document part relationships (`doc.part.rels`).
|
|
147
|
+
- OCR is run before the DOCX→HTML→Markdown pipeline executes: placeholder tokens are injected into the HTML so that the markdown converter does not escape the OCR markers, and the final placeholders are replaced with the formatted `*[Image OCR]...[End OCR]*` blocks after conversion.
|
|
148
|
+
- Document flow (headings, paragraphs, tables) is fully preserved around the OCR blocks.
|
|
149
|
+
|
|
150
|
+
### PPTX
|
|
151
|
+
|
|
152
|
+
- Picture shapes, placeholder shapes with images, and images inside groups are all supported.
|
|
153
|
+
- Shapes are processed in top-to-left reading order per slide.
|
|
154
|
+
- If an `llm_client` is configured, the LLM is asked for a description first; OCR is used as the fallback when no description is returned.
|
|
155
|
+
|
|
156
|
+
### XLSX
|
|
157
|
+
|
|
158
|
+
- Images embedded in worksheets (`sheet._images`) are extracted per sheet.
|
|
159
|
+
- Cell position is calculated from the image anchor coordinates (column/row → Excel letter notation).
|
|
160
|
+
- Images are listed under a `### Images in this sheet:` section after the sheet's data table — they are not interleaved into the table rows.
|
|
161
|
+
|
|
162
|
+
### Output format
|
|
163
|
+
|
|
164
|
+
Every extracted OCR block is wrapped as:
|
|
165
|
+
|
|
166
|
+
```text
|
|
167
|
+
*[Image OCR]
|
|
168
|
+
<extracted text>
|
|
169
|
+
[End OCR]*
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+
## Troubleshooting
|
|
173
|
+
|
|
174
|
+
### OCR text missing from output
|
|
175
|
+
|
|
176
|
+
The most likely cause is a missing `llm_client` or `llm_model`. Verify:
|
|
177
|
+
|
|
178
|
+
```python
|
|
179
|
+
from openai import OpenAI
|
|
180
|
+
from markitdown import MarkItDown
|
|
181
|
+
|
|
182
|
+
md = MarkItDown(
|
|
183
|
+
enable_plugins=True,
|
|
184
|
+
llm_client=OpenAI(), # required
|
|
185
|
+
llm_model="gpt-4o", # required
|
|
186
|
+
)
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
### Plugin not loading
|
|
190
|
+
|
|
191
|
+
Confirm the plugin is installed and discovered:
|
|
192
|
+
|
|
193
|
+
```bash
|
|
194
|
+
markitdown --list-plugins # should show: ocr
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
### API errors
|
|
198
|
+
|
|
199
|
+
The plugin propagates LLM API errors as warnings and continues conversion. Check your API key, quota, and that the chosen model supports vision inputs.
|
|
200
|
+
|
|
201
|
+
## Development
|
|
202
|
+
|
|
203
|
+
### Running Tests
|
|
204
|
+
|
|
205
|
+
```bash
|
|
206
|
+
cd packages/markitdown-ocr
|
|
207
|
+
pytest tests/ -v
|
|
208
|
+
```
|
|
209
|
+
|
|
210
|
+
### Building from Source
|
|
211
|
+
|
|
212
|
+
```bash
|
|
213
|
+
git clone https://github.com/microsoft/markitdown.git
|
|
214
|
+
cd markitdown/packages/markitdown-ocr
|
|
215
|
+
pip install -e .
|
|
216
|
+
```
|
|
217
|
+
|
|
218
|
+
## Contributing
|
|
219
|
+
|
|
220
|
+
Contributions are welcome! See the [MarkItDown repository](https://github.com/microsoft/markitdown) for guidelines.
|
|
221
|
+
|
|
222
|
+
## License
|
|
223
|
+
|
|
224
|
+
MIT — see [LICENSE](LICENSE).
|
|
225
|
+
|
|
226
|
+
## Changelog
|
|
227
|
+
|
|
228
|
+
### 0.1.0 (Initial Release)
|
|
229
|
+
|
|
230
|
+
- LLM Vision OCR for PDF, DOCX, PPTX, XLSX
|
|
231
|
+
- Full-page OCR fallback for scanned PDFs
|
|
232
|
+
- Context-aware inline text insertion
|
|
233
|
+
- Priority-based converter replacement (no code changes required)
|