markitdown-ocr 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,249 @@
1
+ """
2
+ Enhanced PPTX Converter with improved OCR support.
3
+ Already has LLM-based image description, this enhances it with traditional OCR fallback.
4
+ """
5
+
6
+ import io
7
+ import sys
8
+ from typing import Any, BinaryIO, Optional
9
+
10
+ from typing import BinaryIO, Any, Optional
11
+
12
+ from markitdown.converters import HtmlConverter
13
+ from markitdown import DocumentConverter, DocumentConverterResult, StreamInfo
14
+ from markitdown._exceptions import (
15
+ MissingDependencyException,
16
+ MISSING_DEPENDENCY_MESSAGE,
17
+ )
18
+ from ._ocr_service import LLMVisionOCRService
19
+
20
+ _dependency_exc_info = None
21
+ try:
22
+ import pptx
23
+ except ImportError:
24
+ _dependency_exc_info = sys.exc_info()
25
+
26
+
27
+ class PptxConverterWithOCR(DocumentConverter):
28
+ """Enhanced PPTX Converter with OCR fallback."""
29
+
30
+ def __init__(self, ocr_service: Optional[LLMVisionOCRService] = None):
31
+ super().__init__()
32
+ self._html_converter = HtmlConverter()
33
+ self.ocr_service = ocr_service
34
+
35
+ def accepts(
36
+ self,
37
+ file_stream: BinaryIO,
38
+ stream_info: StreamInfo,
39
+ **kwargs: Any,
40
+ ) -> bool:
41
+ mimetype = (stream_info.mimetype or "").lower()
42
+ extension = (stream_info.extension or "").lower()
43
+
44
+ if extension == ".pptx":
45
+ return True
46
+
47
+ if mimetype.startswith(
48
+ "application/vnd.openxmlformats-officedocument.presentationml"
49
+ ):
50
+ return True
51
+
52
+ return False
53
+
54
+ def convert(
55
+ self,
56
+ file_stream: BinaryIO,
57
+ stream_info: StreamInfo,
58
+ **kwargs: Any,
59
+ ) -> DocumentConverterResult:
60
+ if _dependency_exc_info is not None:
61
+ raise MissingDependencyException(
62
+ MISSING_DEPENDENCY_MESSAGE.format(
63
+ converter=type(self).__name__,
64
+ extension=".pptx",
65
+ feature="pptx",
66
+ )
67
+ ) from _dependency_exc_info[1].with_traceback(
68
+ _dependency_exc_info[2]
69
+ ) # type: ignore[union-attr]
70
+
71
+ # Get OCR service (from kwargs or instance)
72
+ ocr_service: Optional[LLMVisionOCRService] = (
73
+ kwargs.get("ocr_service") or self.ocr_service
74
+ )
75
+ llm_client = kwargs.get("llm_client")
76
+
77
+ presentation = pptx.Presentation(file_stream)
78
+ md_content = ""
79
+ slide_num = 0
80
+
81
+ for slide in presentation.slides:
82
+ slide_num += 1
83
+ md_content += f"\\n\\n<!-- Slide number: {slide_num} -->\\n"
84
+
85
+ title = slide.shapes.title
86
+
87
+ def get_shape_content(shape, **kwargs):
88
+ nonlocal md_content
89
+
90
+ # Pictures
91
+ if self._is_picture(shape):
92
+ # Get image data
93
+ image_stream = io.BytesIO(shape.image.blob)
94
+
95
+ # Try LLM description first if available
96
+ llm_description = ""
97
+ if llm_client and kwargs.get("llm_model"):
98
+ try:
99
+ from ._llm_caption import llm_caption
100
+
101
+ image_filename = shape.image.filename
102
+ image_extension = None
103
+ if image_filename:
104
+ import os
105
+
106
+ image_extension = os.path.splitext(image_filename)[1]
107
+
108
+ image_stream_info = StreamInfo(
109
+ mimetype=shape.image.content_type,
110
+ extension=image_extension,
111
+ filename=image_filename,
112
+ )
113
+
114
+ llm_description = llm_caption(
115
+ image_stream,
116
+ image_stream_info,
117
+ client=llm_client,
118
+ model=kwargs.get("llm_model"),
119
+ prompt=kwargs.get("llm_prompt"),
120
+ )
121
+ except Exception:
122
+ pass
123
+
124
+ # Try OCR if LLM failed or not available
125
+ ocr_text = ""
126
+ if not llm_description and ocr_service:
127
+ try:
128
+ image_stream.seek(0)
129
+ ocr_result = ocr_service.extract_text(image_stream)
130
+ if ocr_result.text.strip():
131
+ ocr_text = ocr_result.text.strip()
132
+ except Exception:
133
+ pass
134
+
135
+ # Format extracted content using unified OCR block format
136
+ content = (llm_description or ocr_text or "").strip()
137
+ if content:
138
+ md_content += f"\n*[Image OCR]\n{content}\n[End OCR]*\n"
139
+
140
+ # Tables
141
+ if self._is_table(shape):
142
+ md_content += self._convert_table_to_markdown(shape.table, **kwargs)
143
+
144
+ # Charts
145
+ if shape.has_chart:
146
+ md_content += self._convert_chart_to_markdown(shape.chart)
147
+
148
+ # Text areas
149
+ elif shape.has_text_frame:
150
+ if shape == title:
151
+ md_content += "# " + shape.text.lstrip() + "\\n"
152
+ else:
153
+ md_content += shape.text + "\\n"
154
+
155
+ # Group Shapes
156
+ if shape.shape_type == pptx.enum.shapes.MSO_SHAPE_TYPE.GROUP:
157
+ sorted_shapes = sorted(
158
+ shape.shapes,
159
+ key=lambda x: (
160
+ float("-inf") if not x.top else x.top,
161
+ float("-inf") if not x.left else x.left,
162
+ ),
163
+ )
164
+ for subshape in sorted_shapes:
165
+ get_shape_content(subshape, **kwargs)
166
+
167
+ sorted_shapes = sorted(
168
+ slide.shapes,
169
+ key=lambda x: (
170
+ float("-inf") if not x.top else x.top,
171
+ float("-inf") if not x.left else x.left,
172
+ ),
173
+ )
174
+ for shape in sorted_shapes:
175
+ get_shape_content(shape, **kwargs)
176
+
177
+ md_content = md_content.strip()
178
+
179
+ if slide.has_notes_slide:
180
+ md_content += "\\n\\n### Notes:\\n"
181
+ notes_frame = slide.notes_slide.notes_text_frame
182
+ if notes_frame is not None:
183
+ md_content += notes_frame.text
184
+ md_content = md_content.strip()
185
+
186
+ return DocumentConverterResult(markdown=md_content.strip())
187
+
188
+ def _is_picture(self, shape):
189
+ if shape.shape_type == pptx.enum.shapes.MSO_SHAPE_TYPE.PICTURE:
190
+ return True
191
+ if shape.shape_type == pptx.enum.shapes.MSO_SHAPE_TYPE.PLACEHOLDER:
192
+ if hasattr(shape, "image"):
193
+ return True
194
+ return False
195
+
196
+ def _is_table(self, shape):
197
+ if shape.shape_type == pptx.enum.shapes.MSO_SHAPE_TYPE.TABLE:
198
+ return True
199
+ return False
200
+
201
+ def _convert_table_to_markdown(self, table, **kwargs):
202
+ import html
203
+
204
+ html_table = "<html><body><table>"
205
+ first_row = True
206
+ for row in table.rows:
207
+ html_table += "<tr>"
208
+ for cell in row.cells:
209
+ if first_row:
210
+ html_table += "<th>" + html.escape(cell.text) + "</th>"
211
+ else:
212
+ html_table += "<td>" + html.escape(cell.text) + "</td>"
213
+ html_table += "</tr>"
214
+ first_row = False
215
+ html_table += "</table></body></html>"
216
+
217
+ return (
218
+ self._html_converter.convert_string(html_table, **kwargs).markdown.strip()
219
+ + "\\n"
220
+ )
221
+
222
+ def _convert_chart_to_markdown(self, chart):
223
+ try:
224
+ md = "\\n\\n### Chart"
225
+ if chart.has_title:
226
+ md += f": {chart.chart_title.text_frame.text}"
227
+ md += "\\n\\n"
228
+ data = []
229
+ category_names = [c.label for c in chart.plots[0].categories]
230
+ series_names = [s.name for s in chart.series]
231
+ data.append(["Category"] + series_names)
232
+
233
+ for idx, category in enumerate(category_names):
234
+ row = [category]
235
+ for series in chart.series:
236
+ row.append(series.values[idx])
237
+ data.append(row)
238
+
239
+ markdown_table = []
240
+ for row in data:
241
+ markdown_table.append("| " + " | ".join(map(str, row)) + " |")
242
+ header = markdown_table[0]
243
+ separator = "|" + "|".join(["---"] * len(data[0])) + "|"
244
+ return md + "\\n".join([header, separator] + markdown_table[1:])
245
+ except ValueError as e:
246
+ if "unsupported plot type" in str(e):
247
+ return "\\n\\n[unsupported chart]\\n\\n"
248
+ except Exception:
249
+ return "\\n\\n[unsupported chart]\\n\\n"
@@ -0,0 +1,225 @@
1
+ """
2
+ Enhanced XLSX Converter with OCR support for embedded images.
3
+ Extracts images from Excel spreadsheets and performs OCR while maintaining cell context.
4
+ """
5
+
6
+ import io
7
+ import sys
8
+ from typing import Any, BinaryIO, Optional
9
+
10
+ from markitdown.converters import HtmlConverter
11
+ from markitdown import DocumentConverter, DocumentConverterResult, StreamInfo
12
+ from markitdown._exceptions import (
13
+ MissingDependencyException,
14
+ MISSING_DEPENDENCY_MESSAGE,
15
+ )
16
+ from ._ocr_service import LLMVisionOCRService
17
+
18
+ # Try loading dependencies
19
+ _xlsx_dependency_exc_info = None
20
+ try:
21
+ import pandas as pd
22
+ from openpyxl import load_workbook
23
+ except ImportError:
24
+ _xlsx_dependency_exc_info = sys.exc_info()
25
+
26
+
27
+ class XlsxConverterWithOCR(DocumentConverter):
28
+ """
29
+ Enhanced XLSX Converter with OCR support for embedded images.
30
+ Extracts images with their cell positions and performs OCR.
31
+ """
32
+
33
+ def __init__(self, ocr_service: Optional[LLMVisionOCRService] = None):
34
+ super().__init__()
35
+ self._html_converter = HtmlConverter()
36
+ self.ocr_service = ocr_service
37
+
38
+ def accepts(
39
+ self,
40
+ file_stream: BinaryIO,
41
+ stream_info: StreamInfo,
42
+ **kwargs: Any,
43
+ ) -> bool:
44
+ mimetype = (stream_info.mimetype or "").lower()
45
+ extension = (stream_info.extension or "").lower()
46
+
47
+ if extension == ".xlsx":
48
+ return True
49
+
50
+ if mimetype.startswith(
51
+ "application/vnd.openxmlformats-officedocument.spreadsheetml"
52
+ ):
53
+ return True
54
+
55
+ return False
56
+
57
+ def convert(
58
+ self,
59
+ file_stream: BinaryIO,
60
+ stream_info: StreamInfo,
61
+ **kwargs: Any,
62
+ ) -> DocumentConverterResult:
63
+ if _xlsx_dependency_exc_info is not None:
64
+ raise MissingDependencyException(
65
+ MISSING_DEPENDENCY_MESSAGE.format(
66
+ converter=type(self).__name__,
67
+ extension=".xlsx",
68
+ feature="xlsx",
69
+ )
70
+ ) from _xlsx_dependency_exc_info[1].with_traceback(
71
+ _xlsx_dependency_exc_info[2]
72
+ ) # type: ignore[union-attr]
73
+
74
+ # Get OCR service if available (from kwargs or instance)
75
+ ocr_service: Optional[LLMVisionOCRService] = (
76
+ kwargs.get("ocr_service") or self.ocr_service
77
+ )
78
+
79
+ if ocr_service:
80
+ # Remove ocr_service from kwargs to avoid duplicate argument error
81
+ kwargs_without_ocr = {k: v for k, v in kwargs.items() if k != "ocr_service"}
82
+ return self._convert_with_ocr(
83
+ file_stream, ocr_service, **kwargs_without_ocr
84
+ )
85
+ else:
86
+ return self._convert_standard(file_stream, **kwargs)
87
+
88
+ def _convert_standard(
89
+ self, file_stream: BinaryIO, **kwargs: Any
90
+ ) -> DocumentConverterResult:
91
+ """Standard conversion without OCR."""
92
+ file_stream.seek(0)
93
+ sheets = pd.read_excel(file_stream, sheet_name=None, engine="openpyxl")
94
+ md_content = ""
95
+
96
+ for sheet_name in sheets:
97
+ md_content += f"## {sheet_name}\n"
98
+ html_content = sheets[sheet_name].to_html(index=False)
99
+ md_content += (
100
+ self._html_converter.convert_string(
101
+ html_content, **kwargs
102
+ ).markdown.strip()
103
+ + "\n\n"
104
+ )
105
+
106
+ return DocumentConverterResult(markdown=md_content.strip())
107
+
108
+ def _convert_with_ocr(
109
+ self, file_stream: BinaryIO, ocr_service: LLMVisionOCRService, **kwargs: Any
110
+ ) -> DocumentConverterResult:
111
+ """Convert XLSX with image OCR."""
112
+ file_stream.seek(0)
113
+ wb = load_workbook(file_stream)
114
+
115
+ md_content = ""
116
+
117
+ for sheet_name in wb.sheetnames:
118
+ sheet = wb[sheet_name]
119
+ md_content += f"## {sheet_name}\n\n"
120
+
121
+ # Convert sheet data to markdown table
122
+ file_stream.seek(0)
123
+ try:
124
+ df = pd.read_excel(
125
+ file_stream, sheet_name=sheet_name, engine="openpyxl"
126
+ )
127
+ html_content = df.to_html(index=False)
128
+ md_content += (
129
+ self._html_converter.convert_string(
130
+ html_content, **kwargs
131
+ ).markdown.strip()
132
+ + "\n\n"
133
+ )
134
+ except Exception:
135
+ # If pandas fails, just skip the table
136
+ pass
137
+
138
+ # Extract and OCR images in this sheet
139
+ images_with_ocr = self._extract_and_ocr_sheet_images(sheet, ocr_service)
140
+
141
+ if images_with_ocr:
142
+ md_content += "### Images in this sheet:\n\n"
143
+ for img_info in images_with_ocr:
144
+ ocr_text = img_info["ocr_text"]
145
+ md_content += f"*[Image OCR]\n{ocr_text}\n[End OCR]*\n\n"
146
+
147
+ return DocumentConverterResult(markdown=md_content.strip())
148
+
149
+ def _extract_and_ocr_sheet_images(
150
+ self, sheet: Any, ocr_service: LLMVisionOCRService
151
+ ) -> list[dict]:
152
+ """
153
+ Extract and OCR images from an Excel sheet.
154
+
155
+ Args:
156
+ sheet: openpyxl worksheet
157
+ ocr_service: OCR service
158
+
159
+ Returns:
160
+ List of dicts with 'cell_ref' and 'ocr_text'
161
+ """
162
+ results = []
163
+
164
+ try:
165
+ # Check if sheet has images
166
+ if hasattr(sheet, "_images"):
167
+ for img in sheet._images:
168
+ try:
169
+ # Get image data
170
+ if hasattr(img, "_data"):
171
+ image_data = img._data()
172
+ elif hasattr(img, "image"):
173
+ # Some versions store it differently
174
+ image_data = img.image
175
+ else:
176
+ continue
177
+
178
+ # Create image stream
179
+ image_stream = io.BytesIO(image_data)
180
+
181
+ # Get cell reference
182
+ cell_ref = "unknown"
183
+ if hasattr(img, "anchor"):
184
+ anchor = img.anchor
185
+ if hasattr(anchor, "_from"):
186
+ from_cell = anchor._from
187
+ if hasattr(from_cell, "col") and hasattr(
188
+ from_cell, "row"
189
+ ):
190
+ # Convert column number to letter
191
+ col_letter = self._column_number_to_letter(
192
+ from_cell.col
193
+ )
194
+ cell_ref = f"{col_letter}{from_cell.row + 1}"
195
+
196
+ # Perform OCR
197
+ ocr_result = ocr_service.extract_text(image_stream)
198
+
199
+ if ocr_result.text.strip():
200
+ results.append(
201
+ {
202
+ "cell_ref": cell_ref,
203
+ "ocr_text": ocr_result.text.strip(),
204
+ "backend": ocr_result.backend_used,
205
+ }
206
+ )
207
+
208
+ except Exception:
209
+ continue
210
+
211
+ except Exception:
212
+ pass
213
+
214
+ return results
215
+
216
+ @staticmethod
217
+ def _column_number_to_letter(n: int) -> str:
218
+ """Convert column number to Excel column letter (0-indexed)."""
219
+ result = ""
220
+ n = n + 1 # Make 1-indexed
221
+ while n > 0:
222
+ n -= 1
223
+ result = chr(65 + (n % 26)) + result
224
+ n //= 26
225
+ return result
@@ -0,0 +1,233 @@
1
+ Metadata-Version: 2.4
2
+ Name: markitdown-ocr
3
+ Version: 0.1.0
4
+ Summary: OCR plugin for MarkItDown - Extracts text from images in PDF, DOCX, PPTX, and XLSX via LLM Vision
5
+ Project-URL: Documentation, https://github.com/microsoft/markitdown#readme
6
+ Project-URL: Issues, https://github.com/microsoft/markitdown/issues
7
+ Project-URL: Source, https://github.com/microsoft/markitdown
8
+ Author-email: Contributors <noreply@github.com>
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: docx,llm,markitdown,ocr,pdf,pptx,vision,xlsx
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Programming Language :: Python
14
+ Classifier: Programming Language :: Python :: 3.10
15
+ Classifier: Programming Language :: Python :: 3.11
16
+ Classifier: Programming Language :: Python :: 3.12
17
+ Classifier: Programming Language :: Python :: 3.13
18
+ Classifier: Programming Language :: Python :: Implementation :: CPython
19
+ Requires-Python: >=3.10
20
+ Requires-Dist: mammoth~=1.11.0
21
+ Requires-Dist: markitdown>=0.1.0
22
+ Requires-Dist: openpyxl
23
+ Requires-Dist: pandas
24
+ Requires-Dist: pdfminer-six>=20251230
25
+ Requires-Dist: pdfplumber>=0.11.9
26
+ Requires-Dist: pillow>=9.0.0
27
+ Requires-Dist: pymupdf>=1.24.0
28
+ Requires-Dist: python-docx
29
+ Requires-Dist: python-pptx
30
+ Provides-Extra: llm
31
+ Requires-Dist: openai>=1.0.0; extra == 'llm'
32
+ Description-Content-Type: text/markdown
33
+
34
+ # MarkItDown OCR Plugin
35
+
36
+ LLM Vision plugin for MarkItDown that extracts text from images embedded in PDF, DOCX, PPTX, and XLSX files.
37
+
38
+ Uses the same `llm_client` / `llm_model` pattern that MarkItDown already supports for image descriptions — no new ML libraries or binary dependencies required.
39
+
40
+ ## Features
41
+
42
+ - **Enhanced PDF Converter**: Extracts text from images within PDFs, with full-page OCR fallback for scanned documents
43
+ - **Enhanced DOCX Converter**: OCR for images in Word documents
44
+ - **Enhanced PPTX Converter**: OCR for images in PowerPoint presentations
45
+ - **Enhanced XLSX Converter**: OCR for images in Excel spreadsheets
46
+ - **Context Preservation**: Maintains document structure and flow when inserting extracted text
47
+
48
+ ## Installation
49
+
50
+ ```bash
51
+ pip install markitdown-ocr
52
+ ```
53
+
54
+ The plugin uses whatever OpenAI-compatible client you already have. Install one if you don't have it yet:
55
+
56
+ ```bash
57
+ pip install openai
58
+ ```
59
+
60
+ ## Usage
61
+
62
+ ### Command Line
63
+
64
+ ```bash
65
+ markitdown document.pdf --use-plugins --llm-client openai --llm-model gpt-4o
66
+ ```
67
+
68
+ ### Python API
69
+
70
+ Pass `llm_client` and `llm_model` to `MarkItDown()` exactly as you would for image descriptions:
71
+
72
+ ```python
73
+ from markitdown import MarkItDown
74
+ from openai import OpenAI
75
+
76
+ md = MarkItDown(
77
+ enable_plugins=True,
78
+ llm_client=OpenAI(),
79
+ llm_model="gpt-4o",
80
+ )
81
+
82
+ result = md.convert("document_with_images.pdf")
83
+ print(result.text_content)
84
+ ```
85
+
86
+ If no `llm_client` is provided the plugin still loads, but OCR is silently skipped — falling back to the standard built-in converter.
87
+
88
+ ### Custom Prompt
89
+
90
+ Override the default extraction prompt for specialized documents:
91
+
92
+ ```python
93
+ md = MarkItDown(
94
+ enable_plugins=True,
95
+ llm_client=OpenAI(),
96
+ llm_model="gpt-4o",
97
+ llm_prompt="Extract all text from this image, preserving table structure.",
98
+ )
99
+ ```
100
+
101
+ ### Any OpenAI-Compatible Client
102
+
103
+ Works with any client that follows the OpenAI API:
104
+
105
+ ```python
106
+ from openai import AzureOpenAI
107
+
108
+ md = MarkItDown(
109
+ enable_plugins=True,
110
+ llm_client=AzureOpenAI(
111
+ api_key="...",
112
+ azure_endpoint="https://your-resource.openai.azure.com/",
113
+ api_version="2024-02-01",
114
+ ),
115
+ llm_model="gpt-4o",
116
+ )
117
+ ```
118
+
119
+ ## How It Works
120
+
121
+ When `MarkItDown(enable_plugins=True, llm_client=..., llm_model=...)` is called:
122
+
123
+ 1. MarkItDown discovers the plugin via the `markitdown.plugin` entry point group
124
+ 2. It calls `register_converters()`, forwarding all kwargs including `llm_client` and `llm_model`
125
+ 3. The plugin creates an `LLMVisionOCRService` from those kwargs
126
+ 4. Four OCR-enhanced converters are registered at **priority -1.0** — before the built-in converters at priority 0.0
127
+
128
+ When a file is converted:
129
+
130
+ 1. The OCR converter accepts the file
131
+ 2. It extracts embedded images from the document
132
+ 3. Each image is sent to the LLM with an extraction prompt
133
+ 4. The returned text is inserted inline, preserving document structure
134
+ 5. If the LLM call fails, conversion continues without that image's text
135
+
136
+ ## Supported File Formats
137
+
138
+ ### PDF
139
+
140
+ - Embedded images are extracted by position (via `page.images` / page XObjects) and OCR'd inline, interleaved with the surrounding text in vertical reading order.
141
+ - **Scanned PDFs** (pages with no extractable text) are detected automatically: each page is rendered at 300 DPI and sent to the LLM as a full-page image.
142
+ - **Malformed PDFs** that pdfplumber/pdfminer cannot open (e.g. truncated EOF) are retried with PyMuPDF page rendering, so content is still recovered.
143
+
144
+ ### DOCX
145
+
146
+ - Images are extracted via document part relationships (`doc.part.rels`).
147
+ - OCR is run before the DOCX→HTML→Markdown pipeline executes: placeholder tokens are injected into the HTML so that the markdown converter does not escape the OCR markers, and the final placeholders are replaced with the formatted `*[Image OCR]...[End OCR]*` blocks after conversion.
148
+ - Document flow (headings, paragraphs, tables) is fully preserved around the OCR blocks.
149
+
150
+ ### PPTX
151
+
152
+ - Picture shapes, placeholder shapes with images, and images inside groups are all supported.
153
+ - Shapes are processed in top-to-left reading order per slide.
154
+ - If an `llm_client` is configured, the LLM is asked for a description first; OCR is used as the fallback when no description is returned.
155
+
156
+ ### XLSX
157
+
158
+ - Images embedded in worksheets (`sheet._images`) are extracted per sheet.
159
+ - Cell position is calculated from the image anchor coordinates (column/row → Excel letter notation).
160
+ - Images are listed under a `### Images in this sheet:` section after the sheet's data table — they are not interleaved into the table rows.
161
+
162
+ ### Output format
163
+
164
+ Every extracted OCR block is wrapped as:
165
+
166
+ ```text
167
+ *[Image OCR]
168
+ <extracted text>
169
+ [End OCR]*
170
+ ```
171
+
172
+ ## Troubleshooting
173
+
174
+ ### OCR text missing from output
175
+
176
+ The most likely cause is a missing `llm_client` or `llm_model`. Verify:
177
+
178
+ ```python
179
+ from openai import OpenAI
180
+ from markitdown import MarkItDown
181
+
182
+ md = MarkItDown(
183
+ enable_plugins=True,
184
+ llm_client=OpenAI(), # required
185
+ llm_model="gpt-4o", # required
186
+ )
187
+ ```
188
+
189
+ ### Plugin not loading
190
+
191
+ Confirm the plugin is installed and discovered:
192
+
193
+ ```bash
194
+ markitdown --list-plugins # should show: ocr
195
+ ```
196
+
197
+ ### API errors
198
+
199
+ The plugin propagates LLM API errors as warnings and continues conversion. Check your API key, quota, and that the chosen model supports vision inputs.
200
+
201
+ ## Development
202
+
203
+ ### Running Tests
204
+
205
+ ```bash
206
+ cd packages/markitdown-ocr
207
+ pytest tests/ -v
208
+ ```
209
+
210
+ ### Building from Source
211
+
212
+ ```bash
213
+ git clone https://github.com/microsoft/markitdown.git
214
+ cd markitdown/packages/markitdown-ocr
215
+ pip install -e .
216
+ ```
217
+
218
+ ## Contributing
219
+
220
+ Contributions are welcome! See the [MarkItDown repository](https://github.com/microsoft/markitdown) for guidelines.
221
+
222
+ ## License
223
+
224
+ MIT — see [LICENSE](LICENSE).
225
+
226
+ ## Changelog
227
+
228
+ ### 0.1.0 (Initial Release)
229
+
230
+ - LLM Vision OCR for PDF, DOCX, PPTX, XLSX
231
+ - Full-page OCR fallback for scanned PDFs
232
+ - Context-aware inline text insertion
233
+ - Priority-based converter replacement (no code changes required)