markitdown-ocr 0.1.0__tar.gz → 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/PKG-INFO +24 -12
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/README.md +21 -9
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/pyproject.toml +1 -1
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/src/markitdown_ocr/__about__.py +1 -1
- markitdown_ocr-0.1.1/src/markitdown_ocr/_docx_converter_with_ocr.py +71 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/src/markitdown_ocr/_ocr_service.py +17 -0
- markitdown_ocr-0.1.1/src/markitdown_ocr/_pptx_converter_with_ocr.py +70 -0
- markitdown_ocr-0.1.1/src/markitdown_ocr/_xlsx_converter_with_ocr.py +70 -0
- markitdown_ocr-0.1.1/tests/test_docx_converter.py +382 -0
- markitdown_ocr-0.1.1/tests/test_docx_inheritance.py +287 -0
- markitdown_ocr-0.1.1/tests/test_ocr_metadata.py +134 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/test_pdf_converter.py +19 -6
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/test_pptx_converter.py +67 -24
- markitdown_ocr-0.1.1/tests/test_pptx_inheritance.py +319 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/test_xlsx_converter.py +30 -29
- markitdown_ocr-0.1.1/tests/test_xlsx_inheritance.py +339 -0
- markitdown_ocr-0.1.0/src/markitdown_ocr/_docx_converter_with_ocr.py +0 -189
- markitdown_ocr-0.1.0/src/markitdown_ocr/_pptx_converter_with_ocr.py +0 -249
- markitdown_ocr-0.1.0/src/markitdown_ocr/_xlsx_converter_with_ocr.py +0 -225
- markitdown_ocr-0.1.0/tests/test_docx_converter.py +0 -223
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/.gitignore +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/LICENSE +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/src/markitdown_ocr/__init__.py +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/src/markitdown_ocr/_pdf_converter_with_ocr.py +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/src/markitdown_ocr/_plugin.py +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/__init__.py +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/docx_complex_layout.docx +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/docx_image_end.docx +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/docx_image_middle.docx +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/docx_image_start.docx +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/docx_multipage.docx +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/docx_multiple_images.docx +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_complex_layout.pdf +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_image_end.pdf +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_image_middle.pdf +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_image_start.pdf +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_multipage.pdf +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_multiple_images.pdf +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_scanned_invoice.pdf +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_scanned_meeting_minutes.pdf +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_scanned_minimal.pdf +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_scanned_report.pdf +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_scanned_sales_report.pdf +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pptx_complex_layout.pptx +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pptx_image_end.pptx +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pptx_image_middle.pptx +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pptx_image_start.pptx +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pptx_multiple_images.pptx +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/xlsx_complex_layout.xlsx +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/xlsx_image_end.xlsx +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/xlsx_image_middle.xlsx +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/xlsx_image_start.xlsx +0 -0
- {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/xlsx_multiple_images.xlsx +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: markitdown-ocr
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.1
|
|
4
4
|
Summary: OCR plugin for MarkItDown - Extracts text from images in PDF, DOCX, PPTX, and XLSX via LLM Vision
|
|
5
5
|
Project-URL: Documentation, https://github.com/microsoft/markitdown#readme
|
|
6
6
|
Project-URL: Issues, https://github.com/microsoft/markitdown/issues
|
|
@@ -18,7 +18,7 @@ Classifier: Programming Language :: Python :: 3.13
|
|
|
18
18
|
Classifier: Programming Language :: Python :: Implementation :: CPython
|
|
19
19
|
Requires-Python: >=3.10
|
|
20
20
|
Requires-Dist: mammoth~=1.11.0
|
|
21
|
-
Requires-Dist: markitdown
|
|
21
|
+
Requires-Dist: markitdown<0.2.0,>=0.1.8
|
|
22
22
|
Requires-Dist: openpyxl
|
|
23
23
|
Requires-Dist: pandas
|
|
24
24
|
Requires-Dist: pdfminer-six>=20251230
|
|
@@ -47,6 +47,8 @@ Uses the same `llm_client` / `llm_model` pattern that MarkItDown already support
|
|
|
47
47
|
|
|
48
48
|
## Installation
|
|
49
49
|
|
|
50
|
+
Requires `markitdown>=0.1.8,<0.2.0`, which provides the Office image-rendering hooks used by this plugin. Installing the plugin automatically resolves a compatible core version.
|
|
51
|
+
|
|
50
52
|
```bash
|
|
51
53
|
pip install markitdown-ocr
|
|
52
54
|
```
|
|
@@ -130,9 +132,11 @@ When a file is converted:
|
|
|
130
132
|
1. The OCR converter accepts the file
|
|
131
133
|
2. It extracts embedded images from the document
|
|
132
134
|
3. Each image is sent to the LLM with an extraction prompt
|
|
133
|
-
4. The returned text is
|
|
135
|
+
4. The returned text is placed alongside document content (XLSX images follow their sheet's table)
|
|
134
136
|
5. If the LLM call fails, conversion continues without that image's text
|
|
135
137
|
|
|
138
|
+
The DOCX, PPTX, and XLSX converters subclass their core counterparts and override the same semi-private `_image_to_html` method. Core handles native content, preprocessing, and placement; the plugin supplies escaped OCR HTML, which passes through the shared HTML-to-Markdown renderer. PDF uses its separate existing pipeline.
|
|
139
|
+
|
|
136
140
|
## Supported File Formats
|
|
137
141
|
|
|
138
142
|
### PDF
|
|
@@ -143,21 +147,23 @@ When a file is converted:
|
|
|
143
147
|
|
|
144
148
|
### DOCX
|
|
145
149
|
|
|
146
|
-
-
|
|
147
|
-
-
|
|
148
|
-
-
|
|
150
|
+
- Inherits core DOCX preprocessing, styles, math, and Mammoth conversion.
|
|
151
|
+
- Mammoth provides each embedded image to `_image_to_html`. OCR fragments are inserted into the document's HTML before Markdown rendering, not substituted into finished Markdown.
|
|
152
|
+
- Block fragments split enclosing paragraphs where necessary and remain inside their table cell or list item. Table-cell line breaks follow the shared HTML converter's existing limitations.
|
|
149
153
|
|
|
150
154
|
### PPTX
|
|
151
155
|
|
|
152
156
|
- Picture shapes, placeholder shapes with images, and images inside groups are all supported.
|
|
153
|
-
-
|
|
157
|
+
- Inherits core shape ordering, native text, tables, charts, and speaker notes.
|
|
158
|
+
- Slide content now uses the core converter's real line breaks rather than the old plugin's literal `\n` text, and inherits its empty-title and empty-notes handling.
|
|
154
159
|
- If an `llm_client` is configured, the LLM is asked for a description first; OCR is used as the fallback when no description is returned.
|
|
155
160
|
|
|
156
161
|
### XLSX
|
|
157
162
|
|
|
158
|
-
-
|
|
159
|
-
- Cell position is calculated from the image anchor coordinates (column/row → Excel letter notation).
|
|
163
|
+
- Inherits core workbook repair and table rendering; images are read from the same repaired workbook.
|
|
160
164
|
- Images are listed under a `### Images in this sheet:` section after the sheet's data table — they are not interleaved into the table rows.
|
|
165
|
+
- Sheet heading spacing follows the core converter; no new cell-position labels are added.
|
|
166
|
+
- Legacy `.xls` files remain handled by the existing core converter, without image OCR.
|
|
161
167
|
|
|
162
168
|
### Output format
|
|
163
169
|
|
|
@@ -169,6 +175,10 @@ Every extracted OCR block is wrapped as:
|
|
|
169
175
|
[End OCR]*
|
|
170
176
|
```
|
|
171
177
|
|
|
178
|
+
For Office formats, recognized text is escaped as literal HTML text before Markdown rendering. Markdown escaping and line breaks therefore follow the shared HTML converter: for example, underscores may be backslash-escaped, and direct converter results use Markdown hard breaks. `MarkItDown` subsequently strips trailing whitespace from each output line. Empty recognition retains the native image representation (XLSX normally omits images).
|
|
179
|
+
|
|
180
|
+
Repeated image bytes are recognized once per conversion, while the result is placed at every occurrence. The cache is not shared across documents or service overrides.
|
|
181
|
+
|
|
172
182
|
## Troubleshooting
|
|
173
183
|
|
|
174
184
|
### OCR text missing from output
|
|
@@ -198,6 +208,8 @@ markitdown --list-plugins # should show: ocr
|
|
|
198
208
|
|
|
199
209
|
The plugin propagates LLM API errors as warnings and continues conversion. Check your API key, quota, and that the chosen model supports vision inputs.
|
|
200
210
|
|
|
211
|
+
For Office OCR, a service-reported error emits a warning and retains native image rendering. Exceptions raised by custom OCR services propagate from direct converter calls; `MarkItDown` can retry another applicable converter through its normal fallback behavior.
|
|
212
|
+
|
|
201
213
|
## Development
|
|
202
214
|
|
|
203
215
|
### Running Tests
|
|
@@ -211,8 +223,8 @@ pytest tests/ -v
|
|
|
211
223
|
|
|
212
224
|
```bash
|
|
213
225
|
git clone https://github.com/microsoft/markitdown.git
|
|
214
|
-
cd markitdown
|
|
215
|
-
pip install -e
|
|
226
|
+
cd markitdown
|
|
227
|
+
pip install -e 'packages/markitdown[docx,pptx,xlsx]' -e packages/markitdown-ocr
|
|
216
228
|
```
|
|
217
229
|
|
|
218
230
|
## Contributing
|
|
@@ -14,6 +14,8 @@ Uses the same `llm_client` / `llm_model` pattern that MarkItDown already support
|
|
|
14
14
|
|
|
15
15
|
## Installation
|
|
16
16
|
|
|
17
|
+
Requires `markitdown>=0.1.8,<0.2.0`, which provides the Office image-rendering hooks used by this plugin. Installing the plugin automatically resolves a compatible core version.
|
|
18
|
+
|
|
17
19
|
```bash
|
|
18
20
|
pip install markitdown-ocr
|
|
19
21
|
```
|
|
@@ -97,9 +99,11 @@ When a file is converted:
|
|
|
97
99
|
1. The OCR converter accepts the file
|
|
98
100
|
2. It extracts embedded images from the document
|
|
99
101
|
3. Each image is sent to the LLM with an extraction prompt
|
|
100
|
-
4. The returned text is
|
|
102
|
+
4. The returned text is placed alongside document content (XLSX images follow their sheet's table)
|
|
101
103
|
5. If the LLM call fails, conversion continues without that image's text
|
|
102
104
|
|
|
105
|
+
The DOCX, PPTX, and XLSX converters subclass their core counterparts and override the same semi-private `_image_to_html` method. Core handles native content, preprocessing, and placement; the plugin supplies escaped OCR HTML, which passes through the shared HTML-to-Markdown renderer. PDF uses its separate existing pipeline.
|
|
106
|
+
|
|
103
107
|
## Supported File Formats
|
|
104
108
|
|
|
105
109
|
### PDF
|
|
@@ -110,21 +114,23 @@ When a file is converted:
|
|
|
110
114
|
|
|
111
115
|
### DOCX
|
|
112
116
|
|
|
113
|
-
-
|
|
114
|
-
-
|
|
115
|
-
-
|
|
117
|
+
- Inherits core DOCX preprocessing, styles, math, and Mammoth conversion.
|
|
118
|
+
- Mammoth provides each embedded image to `_image_to_html`. OCR fragments are inserted into the document's HTML before Markdown rendering, not substituted into finished Markdown.
|
|
119
|
+
- Block fragments split enclosing paragraphs where necessary and remain inside their table cell or list item. Table-cell line breaks follow the shared HTML converter's existing limitations.
|
|
116
120
|
|
|
117
121
|
### PPTX
|
|
118
122
|
|
|
119
123
|
- Picture shapes, placeholder shapes with images, and images inside groups are all supported.
|
|
120
|
-
-
|
|
124
|
+
- Inherits core shape ordering, native text, tables, charts, and speaker notes.
|
|
125
|
+
- Slide content now uses the core converter's real line breaks rather than the old plugin's literal `\n` text, and inherits its empty-title and empty-notes handling.
|
|
121
126
|
- If an `llm_client` is configured, the LLM is asked for a description first; OCR is used as the fallback when no description is returned.
|
|
122
127
|
|
|
123
128
|
### XLSX
|
|
124
129
|
|
|
125
|
-
-
|
|
126
|
-
- Cell position is calculated from the image anchor coordinates (column/row → Excel letter notation).
|
|
130
|
+
- Inherits core workbook repair and table rendering; images are read from the same repaired workbook.
|
|
127
131
|
- Images are listed under a `### Images in this sheet:` section after the sheet's data table — they are not interleaved into the table rows.
|
|
132
|
+
- Sheet heading spacing follows the core converter; no new cell-position labels are added.
|
|
133
|
+
- Legacy `.xls` files remain handled by the existing core converter, without image OCR.
|
|
128
134
|
|
|
129
135
|
### Output format
|
|
130
136
|
|
|
@@ -136,6 +142,10 @@ Every extracted OCR block is wrapped as:
|
|
|
136
142
|
[End OCR]*
|
|
137
143
|
```
|
|
138
144
|
|
|
145
|
+
For Office formats, recognized text is escaped as literal HTML text before Markdown rendering. Markdown escaping and line breaks therefore follow the shared HTML converter: for example, underscores may be backslash-escaped, and direct converter results use Markdown hard breaks. `MarkItDown` subsequently strips trailing whitespace from each output line. Empty recognition retains the native image representation (XLSX normally omits images).
|
|
146
|
+
|
|
147
|
+
Repeated image bytes are recognized once per conversion, while the result is placed at every occurrence. The cache is not shared across documents or service overrides.
|
|
148
|
+
|
|
139
149
|
## Troubleshooting
|
|
140
150
|
|
|
141
151
|
### OCR text missing from output
|
|
@@ -165,6 +175,8 @@ markitdown --list-plugins # should show: ocr
|
|
|
165
175
|
|
|
166
176
|
The plugin propagates LLM API errors as warnings and continues conversion. Check your API key, quota, and that the chosen model supports vision inputs.
|
|
167
177
|
|
|
178
|
+
For Office OCR, a service-reported error emits a warning and retains native image rendering. Exceptions raised by custom OCR services propagate from direct converter calls; `MarkItDown` can retry another applicable converter through its normal fallback behavior.
|
|
179
|
+
|
|
168
180
|
## Development
|
|
169
181
|
|
|
170
182
|
### Running Tests
|
|
@@ -178,8 +190,8 @@ pytest tests/ -v
|
|
|
178
190
|
|
|
179
191
|
```bash
|
|
180
192
|
git clone https://github.com/microsoft/markitdown.git
|
|
181
|
-
cd markitdown
|
|
182
|
-
pip install -e
|
|
193
|
+
cd markitdown
|
|
194
|
+
pip install -e 'packages/markitdown[docx,pptx,xlsx]' -e packages/markitdown-ocr
|
|
183
195
|
```
|
|
184
196
|
|
|
185
197
|
## Contributing
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""DOCX image OCR using the core document conversion pipeline."""
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import html
|
|
5
|
+
from typing import Any, BinaryIO, Optional
|
|
6
|
+
from warnings import warn
|
|
7
|
+
|
|
8
|
+
from markitdown import DocumentConverterResult, StreamInfo
|
|
9
|
+
from markitdown.converters import DocxConverter
|
|
10
|
+
|
|
11
|
+
from ._ocr_service import LLMVisionOCRService, _extract_text_with_metadata
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class DocxConverterWithOCR(DocxConverter):
|
|
15
|
+
"""Recognize embedded images while inheriting native DOCX conversion."""
|
|
16
|
+
|
|
17
|
+
def __init__(self, ocr_service: Optional[LLMVisionOCRService] = None):
|
|
18
|
+
super().__init__()
|
|
19
|
+
if not hasattr(DocxConverter, "_image_to_html"):
|
|
20
|
+
raise RuntimeError(
|
|
21
|
+
"DOCX OCR requires markitdown>=0.1.8b3 for the "
|
|
22
|
+
"DocxConverter._image_to_html hook. "
|
|
23
|
+
"Upgrade with: pip install --upgrade 'markitdown>=0.1.8b3'."
|
|
24
|
+
)
|
|
25
|
+
self.ocr_service = ocr_service
|
|
26
|
+
|
|
27
|
+
def convert(
|
|
28
|
+
self,
|
|
29
|
+
file_stream: BinaryIO,
|
|
30
|
+
stream_info: StreamInfo,
|
|
31
|
+
**kwargs: Any,
|
|
32
|
+
) -> DocumentConverterResult:
|
|
33
|
+
# Keep repeated-image recognition local to this document, not the instance.
|
|
34
|
+
kwargs["_docx_ocr_cache"] = {}
|
|
35
|
+
return super().convert(file_stream, stream_info, **kwargs)
|
|
36
|
+
|
|
37
|
+
def _image_to_html(
|
|
38
|
+
self,
|
|
39
|
+
image_stream: BinaryIO,
|
|
40
|
+
stream_info: StreamInfo,
|
|
41
|
+
**kwargs: Any,
|
|
42
|
+
) -> Optional[str]:
|
|
43
|
+
ocr_service = kwargs.get("ocr_service") or self.ocr_service
|
|
44
|
+
if ocr_service is None:
|
|
45
|
+
return None
|
|
46
|
+
|
|
47
|
+
cache: dict[bytes, Optional[str]] = kwargs.get("_docx_ocr_cache", {})
|
|
48
|
+
key = hashlib.sha256(image_stream.read()).digest()
|
|
49
|
+
image_stream.seek(0)
|
|
50
|
+
if key in cache:
|
|
51
|
+
return cache[key]
|
|
52
|
+
|
|
53
|
+
result = _extract_text_with_metadata(ocr_service, image_stream, stream_info)
|
|
54
|
+
if result.error:
|
|
55
|
+
warn(
|
|
56
|
+
f"DOCX image OCR failed: {result.error}. Keeping the native image.",
|
|
57
|
+
RuntimeWarning,
|
|
58
|
+
stacklevel=2,
|
|
59
|
+
)
|
|
60
|
+
cache[key] = None
|
|
61
|
+
return None
|
|
62
|
+
text = result.text.strip()
|
|
63
|
+
if not text:
|
|
64
|
+
cache[key] = None
|
|
65
|
+
return None
|
|
66
|
+
|
|
67
|
+
text = text.replace("\r\n", "\n").replace("\r", "\n")
|
|
68
|
+
content = html.escape(text).replace("\n", "<br>")
|
|
69
|
+
fragment = f"<p><em>[Image OCR]<br>{content}<br>[End OCR]</em></p>"
|
|
70
|
+
cache[key] = fragment
|
|
71
|
+
return fragment
|
|
@@ -4,6 +4,7 @@ Provides LLM Vision-based image text extraction.
|
|
|
4
4
|
"""
|
|
5
5
|
|
|
6
6
|
import base64
|
|
7
|
+
import inspect
|
|
7
8
|
from typing import Any, BinaryIO
|
|
8
9
|
from dataclasses import dataclass
|
|
9
10
|
|
|
@@ -108,3 +109,19 @@ class LLMVisionOCRService:
|
|
|
108
109
|
return OCRResult(text="", backend_used="llm_vision", error=str(e))
|
|
109
110
|
finally:
|
|
110
111
|
image_stream.seek(0)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _extract_text_with_metadata(
|
|
115
|
+
ocr_service: LLMVisionOCRService,
|
|
116
|
+
image_stream: BinaryIO,
|
|
117
|
+
stream_info: StreamInfo,
|
|
118
|
+
) -> OCRResult:
|
|
119
|
+
"""Forward metadata when supported, retaining the legacy stream-only call."""
|
|
120
|
+
extract_text = ocr_service.extract_text
|
|
121
|
+
try:
|
|
122
|
+
inspect.signature(extract_text).bind(image_stream, stream_info=stream_info)
|
|
123
|
+
except (TypeError, ValueError):
|
|
124
|
+
# Unsupported or uninspectable signatures keep the legacy invocation.
|
|
125
|
+
return extract_text(image_stream)
|
|
126
|
+
# Keep service errors outside the signature check; never retry an OCR call.
|
|
127
|
+
return extract_text(image_stream, stream_info=stream_info)
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""PPTX image OCR using the core presentation conversion pipeline."""
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import html
|
|
5
|
+
from typing import Any, BinaryIO, Optional
|
|
6
|
+
from warnings import warn
|
|
7
|
+
|
|
8
|
+
from markitdown import DocumentConverterResult, StreamInfo
|
|
9
|
+
from markitdown.converters import PptxConverter
|
|
10
|
+
|
|
11
|
+
from ._ocr_service import LLMVisionOCRService, _extract_text_with_metadata
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class PptxConverterWithOCR(PptxConverter):
|
|
15
|
+
"""Recognize embedded images while inheriting native PPTX conversion."""
|
|
16
|
+
|
|
17
|
+
def __init__(self, ocr_service: Optional[LLMVisionOCRService] = None):
|
|
18
|
+
super().__init__()
|
|
19
|
+
if not hasattr(PptxConverter, "_image_to_html"):
|
|
20
|
+
raise RuntimeError(
|
|
21
|
+
"PPTX OCR requires markitdown>=0.1.8b3 for the "
|
|
22
|
+
"PptxConverter._image_to_html hook. "
|
|
23
|
+
"Upgrade with: pip install --upgrade 'markitdown>=0.1.8b3'."
|
|
24
|
+
)
|
|
25
|
+
self.ocr_service = ocr_service
|
|
26
|
+
|
|
27
|
+
def convert(
|
|
28
|
+
self,
|
|
29
|
+
file_stream: BinaryIO,
|
|
30
|
+
stream_info: StreamInfo,
|
|
31
|
+
**kwargs: Any,
|
|
32
|
+
) -> DocumentConverterResult:
|
|
33
|
+
kwargs["_pptx_ocr_cache"] = {}
|
|
34
|
+
return super().convert(file_stream, stream_info, **kwargs)
|
|
35
|
+
|
|
36
|
+
def _image_to_html(
|
|
37
|
+
self,
|
|
38
|
+
image_stream: BinaryIO,
|
|
39
|
+
stream_info: StreamInfo,
|
|
40
|
+
**kwargs: Any,
|
|
41
|
+
) -> Optional[str]:
|
|
42
|
+
ocr_service = kwargs.get("ocr_service") or self.ocr_service
|
|
43
|
+
if ocr_service is None:
|
|
44
|
+
return None
|
|
45
|
+
|
|
46
|
+
cache: dict[bytes, Optional[str]] = kwargs.get("_pptx_ocr_cache", {})
|
|
47
|
+
key = hashlib.sha256(image_stream.read()).digest()
|
|
48
|
+
image_stream.seek(0)
|
|
49
|
+
if key in cache:
|
|
50
|
+
return cache[key]
|
|
51
|
+
|
|
52
|
+
result = _extract_text_with_metadata(ocr_service, image_stream, stream_info)
|
|
53
|
+
if result.error:
|
|
54
|
+
warn(
|
|
55
|
+
f"PPTX image OCR failed: {result.error}. Keeping the native image.",
|
|
56
|
+
RuntimeWarning,
|
|
57
|
+
stacklevel=2,
|
|
58
|
+
)
|
|
59
|
+
cache[key] = None
|
|
60
|
+
return None
|
|
61
|
+
text = result.text.strip()
|
|
62
|
+
if not text:
|
|
63
|
+
cache[key] = None
|
|
64
|
+
return None
|
|
65
|
+
|
|
66
|
+
text = text.replace("\r\n", "\n").replace("\r", "\n")
|
|
67
|
+
content = html.escape(text).replace("\n", "<br>")
|
|
68
|
+
fragment = f"<p><em>[Image OCR]<br>{content}<br>[End OCR]</em></p>"
|
|
69
|
+
cache[key] = fragment
|
|
70
|
+
return fragment
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""XLSX image OCR using the core spreadsheet conversion pipeline."""
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import html
|
|
5
|
+
from typing import Any, BinaryIO, Optional
|
|
6
|
+
from warnings import warn
|
|
7
|
+
|
|
8
|
+
from markitdown import DocumentConverterResult, StreamInfo
|
|
9
|
+
from markitdown.converters import XlsxConverter
|
|
10
|
+
|
|
11
|
+
from ._ocr_service import LLMVisionOCRService, _extract_text_with_metadata
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class XlsxConverterWithOCR(XlsxConverter):
|
|
15
|
+
"""Recognize embedded images while inheriting native XLSX conversion."""
|
|
16
|
+
|
|
17
|
+
def __init__(self, ocr_service: Optional[LLMVisionOCRService] = None):
|
|
18
|
+
super().__init__()
|
|
19
|
+
if not hasattr(XlsxConverter, "_image_to_html"):
|
|
20
|
+
raise RuntimeError(
|
|
21
|
+
"XLSX OCR requires markitdown>=0.1.8b3 for the "
|
|
22
|
+
"XlsxConverter._image_to_html hook. "
|
|
23
|
+
"Upgrade with: pip install --upgrade 'markitdown>=0.1.8b3'."
|
|
24
|
+
)
|
|
25
|
+
self.ocr_service = ocr_service
|
|
26
|
+
|
|
27
|
+
def convert(
|
|
28
|
+
self,
|
|
29
|
+
file_stream: BinaryIO,
|
|
30
|
+
stream_info: StreamInfo,
|
|
31
|
+
**kwargs: Any,
|
|
32
|
+
) -> DocumentConverterResult:
|
|
33
|
+
kwargs["_xlsx_ocr_cache"] = {}
|
|
34
|
+
return super().convert(file_stream, stream_info, **kwargs)
|
|
35
|
+
|
|
36
|
+
def _image_to_html(
|
|
37
|
+
self,
|
|
38
|
+
image_stream: BinaryIO,
|
|
39
|
+
stream_info: StreamInfo,
|
|
40
|
+
**kwargs: Any,
|
|
41
|
+
) -> Optional[str]:
|
|
42
|
+
ocr_service = kwargs.get("ocr_service") or self.ocr_service
|
|
43
|
+
if ocr_service is None:
|
|
44
|
+
return None
|
|
45
|
+
|
|
46
|
+
cache: dict[bytes, Optional[str]] = kwargs.get("_xlsx_ocr_cache", {})
|
|
47
|
+
key = hashlib.sha256(image_stream.read()).digest()
|
|
48
|
+
image_stream.seek(0)
|
|
49
|
+
if key in cache:
|
|
50
|
+
return cache[key]
|
|
51
|
+
|
|
52
|
+
result = _extract_text_with_metadata(ocr_service, image_stream, stream_info)
|
|
53
|
+
if result.error:
|
|
54
|
+
warn(
|
|
55
|
+
f"XLSX image OCR failed: {result.error}. Keeping the native image.",
|
|
56
|
+
RuntimeWarning,
|
|
57
|
+
stacklevel=2,
|
|
58
|
+
)
|
|
59
|
+
cache[key] = None
|
|
60
|
+
return None
|
|
61
|
+
text = result.text.strip()
|
|
62
|
+
if not text:
|
|
63
|
+
cache[key] = None
|
|
64
|
+
return None
|
|
65
|
+
|
|
66
|
+
text = text.replace("\r\n", "\n").replace("\r", "\n")
|
|
67
|
+
content = html.escape(text).replace("\n", "<br>")
|
|
68
|
+
fragment = f"<p><em>[Image OCR]<br>{content}<br>[End OCR]</em></p>"
|
|
69
|
+
cache[key] = fragment
|
|
70
|
+
return fragment
|