markitdown-ocr 0.1.0__tar.gz → 0.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/PKG-INFO +24 -12
  2. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/README.md +21 -9
  3. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/pyproject.toml +1 -1
  4. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/src/markitdown_ocr/__about__.py +1 -1
  5. markitdown_ocr-0.1.1/src/markitdown_ocr/_docx_converter_with_ocr.py +71 -0
  6. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/src/markitdown_ocr/_ocr_service.py +17 -0
  7. markitdown_ocr-0.1.1/src/markitdown_ocr/_pptx_converter_with_ocr.py +70 -0
  8. markitdown_ocr-0.1.1/src/markitdown_ocr/_xlsx_converter_with_ocr.py +70 -0
  9. markitdown_ocr-0.1.1/tests/test_docx_converter.py +382 -0
  10. markitdown_ocr-0.1.1/tests/test_docx_inheritance.py +287 -0
  11. markitdown_ocr-0.1.1/tests/test_ocr_metadata.py +134 -0
  12. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/test_pdf_converter.py +19 -6
  13. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/test_pptx_converter.py +67 -24
  14. markitdown_ocr-0.1.1/tests/test_pptx_inheritance.py +319 -0
  15. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/test_xlsx_converter.py +30 -29
  16. markitdown_ocr-0.1.1/tests/test_xlsx_inheritance.py +339 -0
  17. markitdown_ocr-0.1.0/src/markitdown_ocr/_docx_converter_with_ocr.py +0 -189
  18. markitdown_ocr-0.1.0/src/markitdown_ocr/_pptx_converter_with_ocr.py +0 -249
  19. markitdown_ocr-0.1.0/src/markitdown_ocr/_xlsx_converter_with_ocr.py +0 -225
  20. markitdown_ocr-0.1.0/tests/test_docx_converter.py +0 -223
  21. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/.gitignore +0 -0
  22. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/LICENSE +0 -0
  23. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/src/markitdown_ocr/__init__.py +0 -0
  24. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/src/markitdown_ocr/_pdf_converter_with_ocr.py +0 -0
  25. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/src/markitdown_ocr/_plugin.py +0 -0
  26. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/__init__.py +0 -0
  27. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/docx_complex_layout.docx +0 -0
  28. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/docx_image_end.docx +0 -0
  29. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/docx_image_middle.docx +0 -0
  30. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/docx_image_start.docx +0 -0
  31. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/docx_multipage.docx +0 -0
  32. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/docx_multiple_images.docx +0 -0
  33. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_complex_layout.pdf +0 -0
  34. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_image_end.pdf +0 -0
  35. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_image_middle.pdf +0 -0
  36. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_image_start.pdf +0 -0
  37. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_multipage.pdf +0 -0
  38. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_multiple_images.pdf +0 -0
  39. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_scanned_invoice.pdf +0 -0
  40. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_scanned_meeting_minutes.pdf +0 -0
  41. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_scanned_minimal.pdf +0 -0
  42. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_scanned_report.pdf +0 -0
  43. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pdf_scanned_sales_report.pdf +0 -0
  44. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pptx_complex_layout.pptx +0 -0
  45. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pptx_image_end.pptx +0 -0
  46. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pptx_image_middle.pptx +0 -0
  47. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pptx_image_start.pptx +0 -0
  48. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/pptx_multiple_images.pptx +0 -0
  49. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/xlsx_complex_layout.xlsx +0 -0
  50. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/xlsx_image_end.xlsx +0 -0
  51. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/xlsx_image_middle.xlsx +0 -0
  52. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/xlsx_image_start.xlsx +0 -0
  53. {markitdown_ocr-0.1.0 → markitdown_ocr-0.1.1}/tests/ocr_test_data/xlsx_multiple_images.xlsx +0 -0
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: markitdown-ocr
3
- Version: 0.1.0
3
+ Version: 0.1.1
4
4
  Summary: OCR plugin for MarkItDown - Extracts text from images in PDF, DOCX, PPTX, and XLSX via LLM Vision
5
5
  Project-URL: Documentation, https://github.com/microsoft/markitdown#readme
6
6
  Project-URL: Issues, https://github.com/microsoft/markitdown/issues
@@ -18,7 +18,7 @@ Classifier: Programming Language :: Python :: 3.13
18
18
  Classifier: Programming Language :: Python :: Implementation :: CPython
19
19
  Requires-Python: >=3.10
20
20
  Requires-Dist: mammoth~=1.11.0
21
- Requires-Dist: markitdown>=0.1.0
21
+ Requires-Dist: markitdown<0.2.0,>=0.1.8
22
22
  Requires-Dist: openpyxl
23
23
  Requires-Dist: pandas
24
24
  Requires-Dist: pdfminer-six>=20251230
@@ -47,6 +47,8 @@ Uses the same `llm_client` / `llm_model` pattern that MarkItDown already support
47
47
 
48
48
  ## Installation
49
49
 
50
+ Requires `markitdown>=0.1.8,<0.2.0`, which provides the Office image-rendering hooks used by this plugin. Installing the plugin automatically resolves a compatible core version.
51
+
50
52
  ```bash
51
53
  pip install markitdown-ocr
52
54
  ```
@@ -130,9 +132,11 @@ When a file is converted:
130
132
  1. The OCR converter accepts the file
131
133
  2. It extracts embedded images from the document
132
134
  3. Each image is sent to the LLM with an extraction prompt
133
- 4. The returned text is inserted inline, preserving document structure
135
+ 4. The returned text is placed alongside document content (XLSX images follow their sheet's table)
134
136
  5. If the LLM call fails, conversion continues without that image's text
135
137
 
138
+ The DOCX, PPTX, and XLSX converters subclass their core counterparts and override the same semi-private `_image_to_html` method. Core handles native content, preprocessing, and placement; the plugin supplies escaped OCR HTML, which passes through the shared HTML-to-Markdown renderer. PDF uses its separate existing pipeline.
139
+
136
140
  ## Supported File Formats
137
141
 
138
142
  ### PDF
@@ -143,21 +147,23 @@ When a file is converted:
143
147
 
144
148
  ### DOCX
145
149
 
146
- - Images are extracted via document part relationships (`doc.part.rels`).
147
- - OCR is run before the DOCX→HTML→Markdown pipeline executes: placeholder tokens are injected into the HTML so that the markdown converter does not escape the OCR markers, and the final placeholders are replaced with the formatted `*[Image OCR]...[End OCR]*` blocks after conversion.
148
- - Document flow (headings, paragraphs, tables) is fully preserved around the OCR blocks.
150
+ - Inherits core DOCX preprocessing, styles, math, and Mammoth conversion.
151
+ - Mammoth provides each embedded image to `_image_to_html`. OCR fragments are inserted into the document's HTML before Markdown rendering, not substituted into finished Markdown.
152
+ - Block fragments split enclosing paragraphs where necessary and remain inside their table cell or list item. Table-cell line breaks follow the shared HTML converter's existing limitations.
149
153
 
150
154
  ### PPTX
151
155
 
152
156
  - Picture shapes, placeholder shapes with images, and images inside groups are all supported.
153
- - Shapes are processed in top-to-left reading order per slide.
157
+ - Inherits core shape ordering, native text, tables, charts, and speaker notes.
158
+ - Slide content now uses the core converter's real line breaks rather than the old plugin's literal `\n` text, and inherits its empty-title and empty-notes handling.
154
159
  - If an `llm_client` is configured, the LLM is asked for a description first; OCR is used as the fallback when no description is returned.
155
160
 
156
161
  ### XLSX
157
162
 
158
- - Images embedded in worksheets (`sheet._images`) are extracted per sheet.
159
- - Cell position is calculated from the image anchor coordinates (column/row → Excel letter notation).
163
+ - Inherits core workbook repair and table rendering; images are read from the same repaired workbook.
160
164
  - Images are listed under a `### Images in this sheet:` section after the sheet's data table — they are not interleaved into the table rows.
165
+ - Sheet heading spacing follows the core converter; no new cell-position labels are added.
166
+ - Legacy `.xls` files remain handled by the existing core converter, without image OCR.
161
167
 
162
168
  ### Output format
163
169
 
@@ -169,6 +175,10 @@ Every extracted OCR block is wrapped as:
169
175
  [End OCR]*
170
176
  ```
171
177
 
178
+ For Office formats, recognized text is escaped as literal HTML text before Markdown rendering. Markdown escaping and line breaks therefore follow the shared HTML converter: for example, underscores may be backslash-escaped, and direct converter results use Markdown hard breaks. `MarkItDown` subsequently strips trailing whitespace from each output line. Empty recognition retains the native image representation (XLSX normally omits images).
179
+
180
+ Repeated image bytes are recognized once per conversion, while the result is placed at every occurrence. The cache is not shared across documents or service overrides.
181
+
172
182
  ## Troubleshooting
173
183
 
174
184
  ### OCR text missing from output
@@ -198,6 +208,8 @@ markitdown --list-plugins # should show: ocr
198
208
 
199
209
  The plugin propagates LLM API errors as warnings and continues conversion. Check your API key, quota, and that the chosen model supports vision inputs.
200
210
 
211
+ For Office OCR, a service-reported error emits a warning and retains native image rendering. Exceptions raised by custom OCR services propagate from direct converter calls; `MarkItDown` can retry another applicable converter through its normal fallback behavior.
212
+
201
213
  ## Development
202
214
 
203
215
  ### Running Tests
@@ -211,8 +223,8 @@ pytest tests/ -v
211
223
 
212
224
  ```bash
213
225
  git clone https://github.com/microsoft/markitdown.git
214
- cd markitdown/packages/markitdown-ocr
215
- pip install -e .
226
+ cd markitdown
227
+ pip install -e 'packages/markitdown[docx,pptx,xlsx]' -e packages/markitdown-ocr
216
228
  ```
217
229
 
218
230
  ## Contributing
@@ -14,6 +14,8 @@ Uses the same `llm_client` / `llm_model` pattern that MarkItDown already support
14
14
 
15
15
  ## Installation
16
16
 
17
+ Requires `markitdown>=0.1.8,<0.2.0`, which provides the Office image-rendering hooks used by this plugin. Installing the plugin automatically resolves a compatible core version.
18
+
17
19
  ```bash
18
20
  pip install markitdown-ocr
19
21
  ```
@@ -97,9 +99,11 @@ When a file is converted:
97
99
  1. The OCR converter accepts the file
98
100
  2. It extracts embedded images from the document
99
101
  3. Each image is sent to the LLM with an extraction prompt
100
- 4. The returned text is inserted inline, preserving document structure
102
+ 4. The returned text is placed alongside document content (XLSX images follow their sheet's table)
101
103
  5. If the LLM call fails, conversion continues without that image's text
102
104
 
105
+ The DOCX, PPTX, and XLSX converters subclass their core counterparts and override the same semi-private `_image_to_html` method. Core handles native content, preprocessing, and placement; the plugin supplies escaped OCR HTML, which passes through the shared HTML-to-Markdown renderer. PDF uses its separate existing pipeline.
106
+
103
107
  ## Supported File Formats
104
108
 
105
109
  ### PDF
@@ -110,21 +114,23 @@ When a file is converted:
110
114
 
111
115
  ### DOCX
112
116
 
113
- - Images are extracted via document part relationships (`doc.part.rels`).
114
- - OCR is run before the DOCX→HTML→Markdown pipeline executes: placeholder tokens are injected into the HTML so that the markdown converter does not escape the OCR markers, and the final placeholders are replaced with the formatted `*[Image OCR]...[End OCR]*` blocks after conversion.
115
- - Document flow (headings, paragraphs, tables) is fully preserved around the OCR blocks.
117
+ - Inherits core DOCX preprocessing, styles, math, and Mammoth conversion.
118
+ - Mammoth provides each embedded image to `_image_to_html`. OCR fragments are inserted into the document's HTML before Markdown rendering, not substituted into finished Markdown.
119
+ - Block fragments split enclosing paragraphs where necessary and remain inside their table cell or list item. Table-cell line breaks follow the shared HTML converter's existing limitations.
116
120
 
117
121
  ### PPTX
118
122
 
119
123
  - Picture shapes, placeholder shapes with images, and images inside groups are all supported.
120
- - Shapes are processed in top-to-left reading order per slide.
124
+ - Inherits core shape ordering, native text, tables, charts, and speaker notes.
125
+ - Slide content now uses the core converter's real line breaks rather than the old plugin's literal `\n` text, and inherits its empty-title and empty-notes handling.
121
126
  - If an `llm_client` is configured, the LLM is asked for a description first; OCR is used as the fallback when no description is returned.
122
127
 
123
128
  ### XLSX
124
129
 
125
- - Images embedded in worksheets (`sheet._images`) are extracted per sheet.
126
- - Cell position is calculated from the image anchor coordinates (column/row → Excel letter notation).
130
+ - Inherits core workbook repair and table rendering; images are read from the same repaired workbook.
127
131
  - Images are listed under a `### Images in this sheet:` section after the sheet's data table — they are not interleaved into the table rows.
132
+ - Sheet heading spacing follows the core converter; no new cell-position labels are added.
133
+ - Legacy `.xls` files remain handled by the existing core converter, without image OCR.
128
134
 
129
135
  ### Output format
130
136
 
@@ -136,6 +142,10 @@ Every extracted OCR block is wrapped as:
136
142
  [End OCR]*
137
143
  ```
138
144
 
145
+ For Office formats, recognized text is escaped as literal HTML text before Markdown rendering. Markdown escaping and line breaks therefore follow the shared HTML converter: for example, underscores may be backslash-escaped, and direct converter results use Markdown hard breaks. `MarkItDown` subsequently strips trailing whitespace from each output line. Empty recognition retains the native image representation (XLSX normally omits images).
146
+
147
+ Repeated image bytes are recognized once per conversion, while the result is placed at every occurrence. The cache is not shared across documents or service overrides.
148
+
139
149
  ## Troubleshooting
140
150
 
141
151
  ### OCR text missing from output
@@ -165,6 +175,8 @@ markitdown --list-plugins # should show: ocr
165
175
 
166
176
  The plugin propagates LLM API errors as warnings and continues conversion. Check your API key, quota, and that the chosen model supports vision inputs.
167
177
 
178
+ For Office OCR, a service-reported error emits a warning and retains native image rendering. Exceptions raised by custom OCR services propagate from direct converter calls; `MarkItDown` can retry another applicable converter through its normal fallback behavior.
179
+
168
180
  ## Development
169
181
 
170
182
  ### Running Tests
@@ -178,8 +190,8 @@ pytest tests/ -v
178
190
 
179
191
  ```bash
180
192
  git clone https://github.com/microsoft/markitdown.git
181
- cd markitdown/packages/markitdown-ocr
182
- pip install -e .
193
+ cd markitdown
194
+ pip install -e 'packages/markitdown[docx,pptx,xlsx]' -e packages/markitdown-ocr
183
195
  ```
184
196
 
185
197
  ## Contributing
@@ -25,7 +25,7 @@ classifiers = [
25
25
 
26
26
  # Core dependencies — matches the file-format libraries markitdown already uses
27
27
  dependencies = [
28
- "markitdown>=0.1.0",
28
+ "markitdown>=0.1.8,<0.2.0",
29
29
  "pdfminer.six>=20251230",
30
30
  "pdfplumber>=0.11.9",
31
31
  "PyMuPDF>=1.24.0",
@@ -1,4 +1,4 @@
1
1
  # SPDX-FileCopyrightText: 2025-present Contributors
2
2
  # SPDX-License-Identifier: MIT
3
3
 
4
- __version__ = "0.1.0"
4
+ __version__ = "0.1.1"
@@ -0,0 +1,71 @@
1
+ """DOCX image OCR using the core document conversion pipeline."""
2
+
3
+ import hashlib
4
+ import html
5
+ from typing import Any, BinaryIO, Optional
6
+ from warnings import warn
7
+
8
+ from markitdown import DocumentConverterResult, StreamInfo
9
+ from markitdown.converters import DocxConverter
10
+
11
+ from ._ocr_service import LLMVisionOCRService, _extract_text_with_metadata
12
+
13
+
14
+ class DocxConverterWithOCR(DocxConverter):
15
+ """Recognize embedded images while inheriting native DOCX conversion."""
16
+
17
+ def __init__(self, ocr_service: Optional[LLMVisionOCRService] = None):
18
+ super().__init__()
19
+ if not hasattr(DocxConverter, "_image_to_html"):
20
+ raise RuntimeError(
21
+ "DOCX OCR requires markitdown>=0.1.8b3 for the "
22
+ "DocxConverter._image_to_html hook. "
23
+ "Upgrade with: pip install --upgrade 'markitdown>=0.1.8b3'."
24
+ )
25
+ self.ocr_service = ocr_service
26
+
27
+ def convert(
28
+ self,
29
+ file_stream: BinaryIO,
30
+ stream_info: StreamInfo,
31
+ **kwargs: Any,
32
+ ) -> DocumentConverterResult:
33
+ # Keep repeated-image recognition local to this document, not the instance.
34
+ kwargs["_docx_ocr_cache"] = {}
35
+ return super().convert(file_stream, stream_info, **kwargs)
36
+
37
+ def _image_to_html(
38
+ self,
39
+ image_stream: BinaryIO,
40
+ stream_info: StreamInfo,
41
+ **kwargs: Any,
42
+ ) -> Optional[str]:
43
+ ocr_service = kwargs.get("ocr_service") or self.ocr_service
44
+ if ocr_service is None:
45
+ return None
46
+
47
+ cache: dict[bytes, Optional[str]] = kwargs.get("_docx_ocr_cache", {})
48
+ key = hashlib.sha256(image_stream.read()).digest()
49
+ image_stream.seek(0)
50
+ if key in cache:
51
+ return cache[key]
52
+
53
+ result = _extract_text_with_metadata(ocr_service, image_stream, stream_info)
54
+ if result.error:
55
+ warn(
56
+ f"DOCX image OCR failed: {result.error}. Keeping the native image.",
57
+ RuntimeWarning,
58
+ stacklevel=2,
59
+ )
60
+ cache[key] = None
61
+ return None
62
+ text = result.text.strip()
63
+ if not text:
64
+ cache[key] = None
65
+ return None
66
+
67
+ text = text.replace("\r\n", "\n").replace("\r", "\n")
68
+ content = html.escape(text).replace("\n", "<br>")
69
+ fragment = f"<p><em>[Image OCR]<br>{content}<br>[End OCR]</em></p>"
70
+ cache[key] = fragment
71
+ return fragment
@@ -4,6 +4,7 @@ Provides LLM Vision-based image text extraction.
4
4
  """
5
5
 
6
6
  import base64
7
+ import inspect
7
8
  from typing import Any, BinaryIO
8
9
  from dataclasses import dataclass
9
10
 
@@ -108,3 +109,19 @@ class LLMVisionOCRService:
108
109
  return OCRResult(text="", backend_used="llm_vision", error=str(e))
109
110
  finally:
110
111
  image_stream.seek(0)
112
+
113
+
114
+ def _extract_text_with_metadata(
115
+ ocr_service: LLMVisionOCRService,
116
+ image_stream: BinaryIO,
117
+ stream_info: StreamInfo,
118
+ ) -> OCRResult:
119
+ """Forward metadata when supported, retaining the legacy stream-only call."""
120
+ extract_text = ocr_service.extract_text
121
+ try:
122
+ inspect.signature(extract_text).bind(image_stream, stream_info=stream_info)
123
+ except (TypeError, ValueError):
124
+ # Unsupported or uninspectable signatures keep the legacy invocation.
125
+ return extract_text(image_stream)
126
+ # Keep service errors outside the signature check; never retry an OCR call.
127
+ return extract_text(image_stream, stream_info=stream_info)
@@ -0,0 +1,70 @@
1
+ """PPTX image OCR using the core presentation conversion pipeline."""
2
+
3
+ import hashlib
4
+ import html
5
+ from typing import Any, BinaryIO, Optional
6
+ from warnings import warn
7
+
8
+ from markitdown import DocumentConverterResult, StreamInfo
9
+ from markitdown.converters import PptxConverter
10
+
11
+ from ._ocr_service import LLMVisionOCRService, _extract_text_with_metadata
12
+
13
+
14
+ class PptxConverterWithOCR(PptxConverter):
15
+ """Recognize embedded images while inheriting native PPTX conversion."""
16
+
17
+ def __init__(self, ocr_service: Optional[LLMVisionOCRService] = None):
18
+ super().__init__()
19
+ if not hasattr(PptxConverter, "_image_to_html"):
20
+ raise RuntimeError(
21
+ "PPTX OCR requires markitdown>=0.1.8b3 for the "
22
+ "PptxConverter._image_to_html hook. "
23
+ "Upgrade with: pip install --upgrade 'markitdown>=0.1.8b3'."
24
+ )
25
+ self.ocr_service = ocr_service
26
+
27
+ def convert(
28
+ self,
29
+ file_stream: BinaryIO,
30
+ stream_info: StreamInfo,
31
+ **kwargs: Any,
32
+ ) -> DocumentConverterResult:
33
+ kwargs["_pptx_ocr_cache"] = {}
34
+ return super().convert(file_stream, stream_info, **kwargs)
35
+
36
+ def _image_to_html(
37
+ self,
38
+ image_stream: BinaryIO,
39
+ stream_info: StreamInfo,
40
+ **kwargs: Any,
41
+ ) -> Optional[str]:
42
+ ocr_service = kwargs.get("ocr_service") or self.ocr_service
43
+ if ocr_service is None:
44
+ return None
45
+
46
+ cache: dict[bytes, Optional[str]] = kwargs.get("_pptx_ocr_cache", {})
47
+ key = hashlib.sha256(image_stream.read()).digest()
48
+ image_stream.seek(0)
49
+ if key in cache:
50
+ return cache[key]
51
+
52
+ result = _extract_text_with_metadata(ocr_service, image_stream, stream_info)
53
+ if result.error:
54
+ warn(
55
+ f"PPTX image OCR failed: {result.error}. Keeping the native image.",
56
+ RuntimeWarning,
57
+ stacklevel=2,
58
+ )
59
+ cache[key] = None
60
+ return None
61
+ text = result.text.strip()
62
+ if not text:
63
+ cache[key] = None
64
+ return None
65
+
66
+ text = text.replace("\r\n", "\n").replace("\r", "\n")
67
+ content = html.escape(text).replace("\n", "<br>")
68
+ fragment = f"<p><em>[Image OCR]<br>{content}<br>[End OCR]</em></p>"
69
+ cache[key] = fragment
70
+ return fragment
@@ -0,0 +1,70 @@
1
+ """XLSX image OCR using the core spreadsheet conversion pipeline."""
2
+
3
+ import hashlib
4
+ import html
5
+ from typing import Any, BinaryIO, Optional
6
+ from warnings import warn
7
+
8
+ from markitdown import DocumentConverterResult, StreamInfo
9
+ from markitdown.converters import XlsxConverter
10
+
11
+ from ._ocr_service import LLMVisionOCRService, _extract_text_with_metadata
12
+
13
+
14
+ class XlsxConverterWithOCR(XlsxConverter):
15
+ """Recognize embedded images while inheriting native XLSX conversion."""
16
+
17
+ def __init__(self, ocr_service: Optional[LLMVisionOCRService] = None):
18
+ super().__init__()
19
+ if not hasattr(XlsxConverter, "_image_to_html"):
20
+ raise RuntimeError(
21
+ "XLSX OCR requires markitdown>=0.1.8b3 for the "
22
+ "XlsxConverter._image_to_html hook. "
23
+ "Upgrade with: pip install --upgrade 'markitdown>=0.1.8b3'."
24
+ )
25
+ self.ocr_service = ocr_service
26
+
27
+ def convert(
28
+ self,
29
+ file_stream: BinaryIO,
30
+ stream_info: StreamInfo,
31
+ **kwargs: Any,
32
+ ) -> DocumentConverterResult:
33
+ kwargs["_xlsx_ocr_cache"] = {}
34
+ return super().convert(file_stream, stream_info, **kwargs)
35
+
36
+ def _image_to_html(
37
+ self,
38
+ image_stream: BinaryIO,
39
+ stream_info: StreamInfo,
40
+ **kwargs: Any,
41
+ ) -> Optional[str]:
42
+ ocr_service = kwargs.get("ocr_service") or self.ocr_service
43
+ if ocr_service is None:
44
+ return None
45
+
46
+ cache: dict[bytes, Optional[str]] = kwargs.get("_xlsx_ocr_cache", {})
47
+ key = hashlib.sha256(image_stream.read()).digest()
48
+ image_stream.seek(0)
49
+ if key in cache:
50
+ return cache[key]
51
+
52
+ result = _extract_text_with_metadata(ocr_service, image_stream, stream_info)
53
+ if result.error:
54
+ warn(
55
+ f"XLSX image OCR failed: {result.error}. Keeping the native image.",
56
+ RuntimeWarning,
57
+ stacklevel=2,
58
+ )
59
+ cache[key] = None
60
+ return None
61
+ text = result.text.strip()
62
+ if not text:
63
+ cache[key] = None
64
+ return None
65
+
66
+ text = text.replace("\r\n", "\n").replace("\r", "\n")
67
+ content = html.escape(text).replace("\n", "<br>")
68
+ fragment = f"<p><em>[Image OCR]<br>{content}<br>[End OCR]</em></p>"
69
+ cache[key] = fragment
70
+ return fragment