markitdown-pro 1.3.6__tar.gz → 1.3.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/PKG-INFO +1 -1
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/converters/gpt_vision_wrapper.py +19 -2
- markitdown_pro-1.3.7/markitdown_pro/handlers/image_handler.py +84 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/services/openai_services.py +18 -7
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro.egg-info/PKG-INFO +1 -1
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/setup.py +1 -1
- markitdown_pro-1.3.6/markitdown_pro/handlers/image_handler.py +0 -65
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/README.md +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/__init__.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/common/__init__.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/common/isolated_worker.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/common/logger.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/common/schemas.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/common/utils.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/conversion_pipeline.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/converters/__init__.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/converters/azure_doc_intel_wrapper.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/converters/azure_speech_wrapper.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/converters/base.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/converters/markitdown_wrapper.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/converters/pymupdf_wrapper.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/converters/tabular_wrapper.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/converters/unstructured_wrapper.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/converters/youtube_wrapper.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/handlers/__init__.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/handlers/audio_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/handlers/base_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/handlers/email_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/handlers/epub_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/handlers/ipynb_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/handlers/markitdown_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/handlers/markup_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/handlers/office_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/handlers/pdf_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/handlers/pst_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/handlers/tabular_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/handlers/text_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/services/__init__.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/services/azure_doc_intelligence.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/services/azure_speech.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro.egg-info/SOURCES.txt +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro.egg-info/dependency_links.txt +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro.egg-info/requires.txt +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro.egg-info/top_level.txt +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/pyproject.toml +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/setup.cfg +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/tests/conversion_pipeline/__init__.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/tests/conversion_pipeline/test_conversion_pipeline.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/tests/handlers/__init__.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/tests/handlers/test_email_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/tests/handlers/test_epub_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/tests/handlers/test_image_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/tests/handlers/test_ipynb_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/tests/handlers/test_markitdown_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/tests/handlers/test_markup_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/tests/handlers/test_pst_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/tests/handlers/test_tabular_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/tests/handlers/test_text_handler.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/tests/page_count/__init__.py +0 -0
- {markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/tests/page_count/test_pdf_page_count.py +0 -0
{markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/converters/gpt_vision_wrapper.py
RENAMED
|
@@ -63,15 +63,27 @@ class GPTVisionWrapper(ConverterWrapper):
|
|
|
63
63
|
model_name: str = "gpt-4.1-mini",
|
|
64
64
|
api_version: str = "2024-09-01-preview",
|
|
65
65
|
completion_tokens: int = 32000,
|
|
66
|
+
request_timeout_s: float = 20.0,
|
|
67
|
+
acquire_timeout_s: float = 20.0,
|
|
68
|
+
page_timeout_s: float = 20.0,
|
|
66
69
|
):
|
|
67
70
|
super().__init__(model_name)
|
|
68
71
|
self.gpt_vision = GPTVision(
|
|
69
72
|
model_name=model_name,
|
|
70
73
|
api_version=api_version,
|
|
71
74
|
completion_tokens=completion_tokens,
|
|
75
|
+
request_timeout_s=request_timeout_s,
|
|
76
|
+
acquire_timeout_s=acquire_timeout_s,
|
|
77
|
+
page_timeout_s=page_timeout_s,
|
|
72
78
|
)
|
|
73
79
|
|
|
74
|
-
async def convert(
|
|
80
|
+
async def convert(
|
|
81
|
+
self,
|
|
82
|
+
file_path: str,
|
|
83
|
+
max_retries: int = 6,
|
|
84
|
+
base_delay: float = 0.5,
|
|
85
|
+
max_delay: float = 20.0,
|
|
86
|
+
) -> Optional[str]:
|
|
75
87
|
"""
|
|
76
88
|
Convert a file to Markdown using GPT-Vision.
|
|
77
89
|
|
|
@@ -95,7 +107,12 @@ class GPTVisionWrapper(ConverterWrapper):
|
|
|
95
107
|
# Use the concurrent OCR path for PDFs (page-by-page, parallelized).
|
|
96
108
|
return await self.gpt_vision.process_scanned_pdf_concurrent(file_path)
|
|
97
109
|
# Otherwise, assume it's an image and run single-image OCR/analysis.
|
|
98
|
-
return await self.gpt_vision.process_image(
|
|
110
|
+
return await self.gpt_vision.process_image(
|
|
111
|
+
file_or_url=file_path,
|
|
112
|
+
max_retries=max_retries,
|
|
113
|
+
base_delay=base_delay,
|
|
114
|
+
max_delay=max_delay,
|
|
115
|
+
)
|
|
99
116
|
|
|
100
117
|
async def aclose(self) -> None:
|
|
101
118
|
await self.gpt_vision.aclose()
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
import contextlib
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import List, Optional
|
|
4
|
+
|
|
5
|
+
from ..common.logger import logger
|
|
6
|
+
from ..common.utils import detect_extension
|
|
7
|
+
from ..converters.gpt_vision_wrapper import GPTVisionWrapper
|
|
8
|
+
from .base_handler import BaseHandler
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class ImageHandler(BaseHandler):
|
|
12
|
+
SUPPORTED_EXTENSIONS = GPTVisionWrapper.SUPPORTED_EXTENSIONS
|
|
13
|
+
|
|
14
|
+
def __init__(self, *args, **kwargs):
|
|
15
|
+
super().__init__(*args, **kwargs)
|
|
16
|
+
self.gpt_vision_4_1_mini = GPTVisionWrapper(model_name="gpt-4.1-mini")
|
|
17
|
+
self.gpt_vision_4_1 = GPTVisionWrapper(model_name="gpt-4.1")
|
|
18
|
+
|
|
19
|
+
self.pipeline: List[GPTVisionWrapper] = [self.gpt_vision_4_1_mini, self.gpt_vision_4_1]
|
|
20
|
+
|
|
21
|
+
self._max_retries = kwargs.get("max_retries", 2)
|
|
22
|
+
self._base_delay = kwargs.get("base_delay", 0.5)
|
|
23
|
+
self._max_delay = kwargs.get("max_delay", 10.0)
|
|
24
|
+
|
|
25
|
+
async def handle(self, file_path, *args, **kwargs) -> Optional[str]:
|
|
26
|
+
"""
|
|
27
|
+
Handles image files by converting them to Markdown using GPT Vision.
|
|
28
|
+
|
|
29
|
+
Args:
|
|
30
|
+
file_path: Path to the image file.
|
|
31
|
+
|
|
32
|
+
Returns:
|
|
33
|
+
Markdown string representing the image content (OCR and analysis),
|
|
34
|
+
or an error message.
|
|
35
|
+
"""
|
|
36
|
+
for handler in self.pipeline:
|
|
37
|
+
try:
|
|
38
|
+
logger.info(f"ImageHandler: Trying handler {handler.name} for file '{file_path}'")
|
|
39
|
+
md_content = await handler.convert(
|
|
40
|
+
file_path,
|
|
41
|
+
max_retries=self._max_retries,
|
|
42
|
+
base_delay=self._base_delay,
|
|
43
|
+
max_delay=self._max_delay,
|
|
44
|
+
)
|
|
45
|
+
if md_content:
|
|
46
|
+
return md_content
|
|
47
|
+
except Exception as e:
|
|
48
|
+
logger.error(
|
|
49
|
+
f"ImageHandler: Handler {handler.name} failed for file '{file_path}': {e}"
|
|
50
|
+
)
|
|
51
|
+
finally:
|
|
52
|
+
# Prevent "Event loop is closed" from httpx finalizers during pytest teardown
|
|
53
|
+
with contextlib.suppress(Exception):
|
|
54
|
+
if hasattr(handler, "gpt_vision"):
|
|
55
|
+
await handler.gpt_vision.aclose()
|
|
56
|
+
|
|
57
|
+
logger.error(f"ImageHandler: All handlers failed for file '{file_path}'")
|
|
58
|
+
return None
|
|
59
|
+
|
|
60
|
+
async def get_page_count(self, file_path: str | Path) -> Optional[int]:
|
|
61
|
+
"""
|
|
62
|
+
Determines the number of pages in the given image file.
|
|
63
|
+
"""
|
|
64
|
+
p = Path(file_path)
|
|
65
|
+
ext = detect_extension(str(p.absolute()))
|
|
66
|
+
if ext not in self.SUPPORTED_EXTENSIONS:
|
|
67
|
+
logger.warning(f"ImageHandler: Unsupported file format: {ext}")
|
|
68
|
+
return None
|
|
69
|
+
|
|
70
|
+
if ext in {".tif", ".tiff"}:
|
|
71
|
+
try:
|
|
72
|
+
from PIL import Image, ImageSequence # optional dependency
|
|
73
|
+
|
|
74
|
+
with Image.open(str(p)) as im:
|
|
75
|
+
return sum(1 for _ in ImageSequence.Iterator(im)) or 1
|
|
76
|
+
except Exception as e_tiff:
|
|
77
|
+
logger.warning(f"ImageHandler: local TIFF page count failed for '{p}': {e_tiff}")
|
|
78
|
+
else:
|
|
79
|
+
return 1
|
|
80
|
+
|
|
81
|
+
async def aclose(self) -> None:
|
|
82
|
+
for handler in self.pipeline:
|
|
83
|
+
if hasattr(handler, "gpt_vision") and handler.gpt_vision:
|
|
84
|
+
await handler.gpt_vision.aclose()
|
|
@@ -71,9 +71,9 @@ class GPTVision:
|
|
|
71
71
|
completion_tokens: int,
|
|
72
72
|
max_concurrency: int = 150,
|
|
73
73
|
*,
|
|
74
|
-
request_timeout_s: float =
|
|
75
|
-
acquire_timeout_s: float =
|
|
76
|
-
page_timeout_s: float =
|
|
74
|
+
request_timeout_s: float = 20.0,
|
|
75
|
+
acquire_timeout_s: float = 20.0,
|
|
76
|
+
page_timeout_s: float = 20.0,
|
|
77
77
|
):
|
|
78
78
|
"""
|
|
79
79
|
Parameters
|
|
@@ -326,7 +326,14 @@ class GPTVision:
|
|
|
326
326
|
# Public OCR APIs
|
|
327
327
|
# -------------------------
|
|
328
328
|
|
|
329
|
-
async def process_image(
|
|
329
|
+
async def process_image(
|
|
330
|
+
self,
|
|
331
|
+
file_or_url: str,
|
|
332
|
+
prompt: Optional[str] = None,
|
|
333
|
+
max_retries: int = 6,
|
|
334
|
+
base_delay: float = 0.5,
|
|
335
|
+
max_delay: float = 20.0,
|
|
336
|
+
) -> Optional[str]:
|
|
330
337
|
"""
|
|
331
338
|
OCR a single image (local path or URL) with the vision model.
|
|
332
339
|
|
|
@@ -363,7 +370,11 @@ class GPTVision:
|
|
|
363
370
|
try:
|
|
364
371
|
# The LLM call itself has its own hard timeout & retry
|
|
365
372
|
response = await self._ainvoke_with_retry(
|
|
366
|
-
file_or_url,
|
|
373
|
+
file_or_url,
|
|
374
|
+
messages=[system_message, human_message],
|
|
375
|
+
max_retries=max_retries,
|
|
376
|
+
base_delay=base_delay,
|
|
377
|
+
max_delay=max_delay,
|
|
367
378
|
)
|
|
368
379
|
finally:
|
|
369
380
|
with suppress(Exception):
|
|
@@ -378,7 +389,7 @@ class GPTVision:
|
|
|
378
389
|
|
|
379
390
|
except asyncio.TimeoutError:
|
|
380
391
|
logger.error(f"GPTVision: process_image: Timed out for {file_or_url}")
|
|
381
|
-
return
|
|
392
|
+
return None
|
|
382
393
|
except Exception as e:
|
|
383
394
|
logger.error(
|
|
384
395
|
f"GPTVision: process_image: Error during GPT-vision image OCR for {file_or_url}: {e}"
|
|
@@ -466,7 +477,7 @@ class GPTVision:
|
|
|
466
477
|
logger.error(
|
|
467
478
|
f"GPTVission: ocr_page: {file_stem} - page {page_index} timed out after {self.page_timeout_s}s"
|
|
468
479
|
)
|
|
469
|
-
partial_md =
|
|
480
|
+
partial_md = None
|
|
470
481
|
|
|
471
482
|
if partial_md:
|
|
472
483
|
logger.info(
|
|
@@ -1,65 +0,0 @@
|
|
|
1
|
-
import contextlib
|
|
2
|
-
from pathlib import Path
|
|
3
|
-
from typing import Optional
|
|
4
|
-
|
|
5
|
-
from ..common.logger import logger
|
|
6
|
-
from ..common.utils import detect_extension
|
|
7
|
-
from ..converters.gpt_vision_wrapper import GPTVisionWrapper
|
|
8
|
-
from .base_handler import BaseHandler
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
class ImageHandler(BaseHandler):
|
|
12
|
-
SUPPORTED_EXTENSIONS = GPTVisionWrapper.SUPPORTED_EXTENSIONS
|
|
13
|
-
|
|
14
|
-
def __init__(self, *args, **kwargs):
|
|
15
|
-
super().__init__(*args, **kwargs)
|
|
16
|
-
self.gpt_vision = GPTVisionWrapper()
|
|
17
|
-
|
|
18
|
-
async def handle(self, file_path, *args, **kwargs) -> Optional[str]:
|
|
19
|
-
"""
|
|
20
|
-
Handles image files by converting them to Markdown using GPT Vision.
|
|
21
|
-
|
|
22
|
-
Args:
|
|
23
|
-
file_path: Path to the image file.
|
|
24
|
-
|
|
25
|
-
Returns:
|
|
26
|
-
Markdown string representing the image content (OCR and analysis),
|
|
27
|
-
or an error message.
|
|
28
|
-
"""
|
|
29
|
-
logger.info(f"Processing image file: {file_path}")
|
|
30
|
-
try:
|
|
31
|
-
md_content = await self.gpt_vision.convert(file_path)
|
|
32
|
-
return md_content
|
|
33
|
-
except Exception as e:
|
|
34
|
-
logger.error(f"ImageHandler: Error handling image file '{file_path}': {e}")
|
|
35
|
-
return None
|
|
36
|
-
finally:
|
|
37
|
-
# Prevent "Event loop is closed" from httpx finalizers during pytest teardown
|
|
38
|
-
with contextlib.suppress(Exception):
|
|
39
|
-
if hasattr(self.gpt_vision, "gpt_vision"):
|
|
40
|
-
await self.gpt_vision.gpt_vision.aclose()
|
|
41
|
-
|
|
42
|
-
async def get_page_count(self, file_path: str | Path) -> Optional[int]:
|
|
43
|
-
"""
|
|
44
|
-
Determines the number of pages in the given image file.
|
|
45
|
-
"""
|
|
46
|
-
p = Path(file_path)
|
|
47
|
-
ext = detect_extension(str(p.absolute()))
|
|
48
|
-
if ext not in self.SUPPORTED_EXTENSIONS:
|
|
49
|
-
logger.warning(f"ImageHandler: Unsupported file format: {ext}")
|
|
50
|
-
return None
|
|
51
|
-
|
|
52
|
-
if ext in {".tif", ".tiff"}:
|
|
53
|
-
try:
|
|
54
|
-
from PIL import Image, ImageSequence # optional dependency
|
|
55
|
-
|
|
56
|
-
with Image.open(str(p)) as im:
|
|
57
|
-
return sum(1 for _ in ImageSequence.Iterator(im)) or 1
|
|
58
|
-
except Exception as e_tiff:
|
|
59
|
-
logger.warning(f"ImageHandler: local TIFF page count failed for '{p}': {e_tiff}")
|
|
60
|
-
else:
|
|
61
|
-
return 1
|
|
62
|
-
|
|
63
|
-
async def aclose(self) -> None:
|
|
64
|
-
if hasattr(self, "gpt_vision") and self.gpt_vision:
|
|
65
|
-
await self.gpt_vision.aclose()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/converters/azure_doc_intel_wrapper.py
RENAMED
|
File without changes
|
{markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/converters/azure_speech_wrapper.py
RENAMED
|
File without changes
|
|
File without changes
|
{markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/converters/markitdown_wrapper.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/converters/unstructured_wrapper.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/markitdown_pro/services/azure_doc_intelligence.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{markitdown_pro-1.3.6 → markitdown_pro-1.3.7}/tests/conversion_pipeline/test_conversion_pipeline.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|