polytext 0.2.8b4__tar.gz → 0.2.8b5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {polytext-0.2.8b4 → polytext-0.2.8b5}/PKG-INFO +1 -1
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/converter/document_ocr_to_text.py +14 -2
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/base.py +6 -1
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/document_ocr.py +5 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext.egg-info/PKG-INFO +1 -1
- {polytext-0.2.8b4 → polytext-0.2.8b5}/setup.py +1 -1
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_ocr_fallbacks.py +41 -1
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_ocr_image_descriptions.py +17 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/LICENSE +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/README.md +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/__init__.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/converter/__init__.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/converter/audio_to_text.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/converter/base.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/converter/beautiful_text.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/converter/document_ocr_to_text_azure_oai.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/converter/gemini_quality_guards.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/converter/html_to_md.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/converter/md_to_text.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/converter/ocr_to_text.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/converter/ocr_to_text_azure_oai.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/converter/pdf.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/converter/text_to_md.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/converter/video_to_audio.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/exceptions/__init__.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/exceptions/base.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/generator/__init__.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/generator/pdf.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/__init__.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/audio.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/aws_auth.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/document.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/downloader/__init__.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/downloader/downloader.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/html.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/markdown.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/notebook.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/ocr.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/plain_text.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/video.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/xml_xbrl.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/youtube.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/loader/youtube_llm.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/processor/__init__.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/processor/audio_chunker.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/processor/text_merger.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/processor/transcript_chunker.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/prompts/__init__.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/prompts/beautiful_text.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/prompts/ocr.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/prompts/text_merging.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/prompts/text_to_md.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/prompts/transcription.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/utils/__init__.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext/utils/utils.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext.egg-info/SOURCES.txt +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext.egg-info/dependency_links.txt +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext.egg-info/not-zip-safe +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext.egg-info/requires.txt +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/polytext.egg-info/top_level.txt +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/pyproject.toml +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/setup.cfg +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_audio_chunker.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_audio_comparison_helpers.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_audio_transcription_model_migration.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_aws_auth.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_base_loader_error_mapping.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_beautiful_text_manual.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_compare_audio_models.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_compare_document_ocr_to_text_models.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_compare_ocr_to_text_models.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_compare_youtube_models.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_dowload_audio_from_youtube.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_dowload_audio_from_youtube_helpers.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_extracted_text_whitespace.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_gemini_quality_guards.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_get_audio_transcript_from_gcs.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_get_customized_pdf_from_markdown.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_get_document_ocr.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_get_document_ocr_azure_oai.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_get_document_text.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_get_document_text_from_gcs.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_get_ocr_from_image.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_get_text_from_markdown.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_get_video_transcript_from_gcs.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_library.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_markdown_loader_gzip.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_markitdown_html.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_notebook_loader.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_pain_text.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_pdf_conversion_error.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_python_version_metadata.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_split_audio_with_llm.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_transcribe_s3_images_from_csv.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_transcribe_s3_images_from_csv_script.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_xml_xbrl_loader.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_youtube_gemini_minimal_check.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_youtube_llm_fallbacks.py +0 -0
- {polytext-0.2.8b4 → polytext-0.2.8b5}/tests/test_youtube_transcript.py +0 -0
|
@@ -115,6 +115,7 @@ def get_document_ocr(
|
|
|
115
115
|
max_output_tokens: int | None = None,
|
|
116
116
|
include_image_descriptions: bool = False,
|
|
117
117
|
allow_partial_ocr_failures: bool = False,
|
|
118
|
+
ocr_render_dpi: int | None = None,
|
|
118
119
|
):
|
|
119
120
|
"""
|
|
120
121
|
Convenience function to extract text from an image file using OCR, optionally formatted as Markdown.
|
|
@@ -141,6 +142,8 @@ def get_document_ocr(
|
|
|
141
142
|
allow_partial_ocr_failures (bool, optional): If True, pages that still
|
|
142
143
|
fail OCR after all retries are recorded inline instead of aborting
|
|
143
144
|
the whole document extraction. Defaults to False.
|
|
145
|
+
ocr_render_dpi (int | None, optional): Render PDF pages at this DPI before
|
|
146
|
+
OCR. When omitted, preserves PyMuPDF's default rendering behavior.
|
|
144
147
|
|
|
145
148
|
Returns:
|
|
146
149
|
dict: Dictionary containing the OCR results and metadata.
|
|
@@ -155,6 +158,7 @@ def get_document_ocr(
|
|
|
155
158
|
max_output_tokens=max_output_tokens,
|
|
156
159
|
include_image_descriptions=include_image_descriptions,
|
|
157
160
|
allow_partial_ocr_failures=allow_partial_ocr_failures,
|
|
161
|
+
ocr_render_dpi=ocr_render_dpi,
|
|
158
162
|
)
|
|
159
163
|
return converter.get_document_ocr(document_for_ocr)
|
|
160
164
|
|
|
@@ -164,7 +168,8 @@ class DocumentOCRToTextConverter:
|
|
|
164
168
|
page_range=None, timeout_minutes: int = None, fallback_stage: int = 0,
|
|
165
169
|
max_output_tokens: int | None = None, include_image_descriptions: bool = False,
|
|
166
170
|
prompt_variant: str = OCR_PROMPT_VARIANT_DEFAULT,
|
|
167
|
-
allow_partial_ocr_failures: bool = False
|
|
171
|
+
allow_partial_ocr_failures: bool = False,
|
|
172
|
+
ocr_render_dpi: int | None = None):
|
|
168
173
|
"""
|
|
169
174
|
Initialize the DocumentOCRToTextConverter class with specified OCR model and formatting options.
|
|
170
175
|
|
|
@@ -192,6 +197,8 @@ class DocumentOCRToTextConverter:
|
|
|
192
197
|
allow_partial_ocr_failures (bool, optional): If True, pages that still
|
|
193
198
|
fail OCR after all retries are recorded inline instead of aborting
|
|
194
199
|
the whole document extraction. Defaults to False.
|
|
200
|
+
ocr_render_dpi (int | None, optional): Render PDF pages at this DPI
|
|
201
|
+
before OCR. Defaults to PyMuPDF's native rendering behavior.
|
|
195
202
|
|
|
196
203
|
Raises:
|
|
197
204
|
OSError: If temp directory creation fails
|
|
@@ -207,6 +214,7 @@ class DocumentOCRToTextConverter:
|
|
|
207
214
|
self.include_image_descriptions = include_image_descriptions
|
|
208
215
|
self.prompt_variant = prompt_variant
|
|
209
216
|
self.allow_partial_ocr_failures = allow_partial_ocr_failures
|
|
217
|
+
self.ocr_render_dpi = ocr_render_dpi
|
|
210
218
|
requested_output_tokens = OCR_MAX_OUTPUT_TOKENS if max_output_tokens is None else max_output_tokens
|
|
211
219
|
self.max_output_tokens = max(requested_output_tokens, OCR_MIN_OUTPUT_TOKENS)
|
|
212
220
|
self.fallback_stage = fallback_stage
|
|
@@ -297,6 +305,7 @@ class DocumentOCRToTextConverter:
|
|
|
297
305
|
include_image_descriptions=self.include_image_descriptions,
|
|
298
306
|
prompt_variant=resolved_prompt_variant,
|
|
299
307
|
allow_partial_ocr_failures=self.allow_partial_ocr_failures,
|
|
308
|
+
ocr_render_dpi=self.ocr_render_dpi,
|
|
300
309
|
)
|
|
301
310
|
result = fallback_converter.get_ocr(
|
|
302
311
|
file_for_ocr=file_for_ocr,
|
|
@@ -553,7 +562,10 @@ class DocumentOCRToTextConverter:
|
|
|
553
562
|
|
|
554
563
|
try:
|
|
555
564
|
# Convert page to image
|
|
556
|
-
|
|
565
|
+
if self.ocr_render_dpi is None:
|
|
566
|
+
pix = page.get_pixmap()
|
|
567
|
+
else:
|
|
568
|
+
pix = page.get_pixmap(dpi=self.ocr_render_dpi, alpha=False)
|
|
557
569
|
pix.save(temp_image_path)
|
|
558
570
|
|
|
559
571
|
# Perform OCR on the page
|
|
@@ -92,6 +92,7 @@ class BaseLoader:
|
|
|
92
92
|
ocr_model: str = "gpt-5-mini", timeout_minutes: int | None = None,
|
|
93
93
|
include_image_descriptions: bool | None = None,
|
|
94
94
|
force_ocr: bool = False,
|
|
95
|
+
ocr_render_dpi: int | None = None,
|
|
95
96
|
**kwargs):
|
|
96
97
|
"""
|
|
97
98
|
Initialize the BaseLoader with cloud storage and LLM configurations.
|
|
@@ -112,6 +113,8 @@ class BaseLoader:
|
|
|
112
113
|
If None, defaults from OCR_INCLUDE_IMAGE_DESCRIPTIONS. Defaults to None.
|
|
113
114
|
force_ocr (bool, optional): If True, supported document files are routed to
|
|
114
115
|
OCRLoader instead of the standard DocumentLoader. Defaults to False.
|
|
116
|
+
ocr_render_dpi (int | None, optional): Render document pages at this DPI
|
|
117
|
+
before OCR. Defaults to the existing renderer behavior.
|
|
115
118
|
**kwargs: Additional keyword arguments to pass to the underlying loader or extraction logic.
|
|
116
119
|
- target_size (int, optional): Target file size in bytes. Defaults to 1MB
|
|
117
120
|
- source (str): Source of the document. Must be either "cloud" or "local"
|
|
@@ -136,6 +139,7 @@ class BaseLoader:
|
|
|
136
139
|
else include_image_descriptions
|
|
137
140
|
)
|
|
138
141
|
self.force_ocr = force_ocr
|
|
142
|
+
self.ocr_render_dpi = ocr_render_dpi
|
|
139
143
|
self.kwargs = kwargs
|
|
140
144
|
self.target_size = kwargs.get("target_size", 1)
|
|
141
145
|
self.source = kwargs.get("source", "cloud")
|
|
@@ -422,7 +426,7 @@ class BaseLoader:
|
|
|
422
426
|
file_extension = file_extension.lower()
|
|
423
427
|
|
|
424
428
|
if is_document_fallback:
|
|
425
|
-
return DocumentOCRLoader(llm_api_key=llm_api_key, markdown_output=self.markdown_output, temp_dir=self.temp_dir, timeout_minutes=self.timeout_minutes, ocr_provider=self.provider, ocr_model=self.ocr_model, include_image_descriptions=self.include_image_descriptions, **kwargs)
|
|
429
|
+
return DocumentOCRLoader(llm_api_key=llm_api_key, markdown_output=self.markdown_output, temp_dir=self.temp_dir, timeout_minutes=self.timeout_minutes, ocr_provider=self.provider, ocr_model=self.ocr_model, include_image_descriptions=self.include_image_descriptions, ocr_render_dpi=self.ocr_render_dpi, **kwargs)
|
|
426
430
|
|
|
427
431
|
if file_extension in [".xml", ".xbrl"]:
|
|
428
432
|
return XmlXbrlLoader(temp_dir=self.temp_dir, markdown_output=self.markdown_output, **kwargs)
|
|
@@ -458,6 +462,7 @@ class BaseLoader:
|
|
|
458
462
|
ocr_provider=self.provider,
|
|
459
463
|
ocr_model=self.ocr_model,
|
|
460
464
|
include_image_descriptions=self.include_image_descriptions,
|
|
465
|
+
ocr_render_dpi=self.ocr_render_dpi,
|
|
461
466
|
**document_kwargs,
|
|
462
467
|
)
|
|
463
468
|
|
|
@@ -47,6 +47,7 @@ class DocumentOCRLoader:
|
|
|
47
47
|
ocr_model: str | None = None,
|
|
48
48
|
include_image_descriptions: bool = False,
|
|
49
49
|
allow_partial_ocr_failures: bool = False,
|
|
50
|
+
ocr_render_dpi: int | None = None,
|
|
50
51
|
**kwargs
|
|
51
52
|
):
|
|
52
53
|
"""
|
|
@@ -86,6 +87,8 @@ class DocumentOCRLoader:
|
|
|
86
87
|
allow_partial_ocr_failures (bool, optional): If True, pages that still
|
|
87
88
|
fail OCR after all retries are recorded inline instead of aborting
|
|
88
89
|
the whole document extraction. Defaults to False.
|
|
90
|
+
ocr_render_dpi (int | None, optional): Render PDF pages at this DPI
|
|
91
|
+
before OCR. Defaults to the existing PyMuPDF rendering behavior.
|
|
89
92
|
**kwargs:
|
|
90
93
|
max_output_tokens (int, optional): Maximum Gemini output tokens for
|
|
91
94
|
Google document OCR generation.
|
|
@@ -110,6 +113,7 @@ class DocumentOCRLoader:
|
|
|
110
113
|
self.ocr_model = ocr_model
|
|
111
114
|
self.include_image_descriptions = include_image_descriptions
|
|
112
115
|
self.allow_partial_ocr_failures = allow_partial_ocr_failures
|
|
116
|
+
self.ocr_render_dpi = ocr_render_dpi
|
|
113
117
|
self.max_output_tokens = kwargs.get("max_output_tokens")
|
|
114
118
|
|
|
115
119
|
# Set up custom temp directory
|
|
@@ -277,6 +281,7 @@ class DocumentOCRLoader:
|
|
|
277
281
|
max_output_tokens=self.max_output_tokens,
|
|
278
282
|
include_image_descriptions=self.include_image_descriptions,
|
|
279
283
|
allow_partial_ocr_failures=self.allow_partial_ocr_failures,
|
|
284
|
+
ocr_render_dpi=self.ocr_render_dpi,
|
|
280
285
|
)
|
|
281
286
|
|
|
282
287
|
result_dict["type"] = self.type
|
|
@@ -51,7 +51,7 @@ def get_requirements(*requirements_file):
|
|
|
51
51
|
|
|
52
52
|
setup(
|
|
53
53
|
name='polytext',
|
|
54
|
-
version='0.2.
|
|
54
|
+
version='0.2.8b5',
|
|
55
55
|
url='https://github.com/docsity/polytext',
|
|
56
56
|
# download_url='https://github.com/pualien/py-polytext/archive/0.1.23.tar.gz',
|
|
57
57
|
license='MIT',
|
|
@@ -88,8 +88,10 @@ class _FakePixmap:
|
|
|
88
88
|
class _FakePage:
|
|
89
89
|
def __init__(self, payload: bytes = b"fake-page-image"):
|
|
90
90
|
self.payload = payload
|
|
91
|
+
self.pixmap_calls = []
|
|
91
92
|
|
|
92
|
-
def get_pixmap(self):
|
|
93
|
+
def get_pixmap(self, **kwargs):
|
|
94
|
+
self.pixmap_calls.append(kwargs)
|
|
93
95
|
return _FakePixmap(payload=self.payload)
|
|
94
96
|
|
|
95
97
|
|
|
@@ -112,6 +114,44 @@ def _immediate_as_completed(futures):
|
|
|
112
114
|
|
|
113
115
|
|
|
114
116
|
class TestOcrFallbacks(unittest.TestCase):
|
|
117
|
+
@patch("polytext.converter.document_ocr_to_text.genai.Client")
|
|
118
|
+
@patch("concurrent.futures.as_completed", side_effect=_immediate_as_completed)
|
|
119
|
+
@patch("concurrent.futures.ThreadPoolExecutor", _ImmediateExecutor)
|
|
120
|
+
@patch("fitz.open")
|
|
121
|
+
def test_document_ocr_renders_pages_at_requested_dpi(
|
|
122
|
+
self,
|
|
123
|
+
mock_fitz_open,
|
|
124
|
+
_mock_as_completed,
|
|
125
|
+
mock_client_cls,
|
|
126
|
+
):
|
|
127
|
+
page = _FakePage()
|
|
128
|
+
mock_fitz_open.return_value = _FakePdf([page])
|
|
129
|
+
mock_client_cls.return_value = _FakeClient()
|
|
130
|
+
|
|
131
|
+
converter = DocumentOCRToTextConverter(ocr_render_dpi=200)
|
|
132
|
+
converter.get_document_ocr("dummy.pdf")
|
|
133
|
+
|
|
134
|
+
self.assertEqual(page.pixmap_calls, [{"dpi": 200, "alpha": False}])
|
|
135
|
+
|
|
136
|
+
@patch("polytext.converter.document_ocr_to_text.genai.Client")
|
|
137
|
+
@patch("concurrent.futures.as_completed", side_effect=_immediate_as_completed)
|
|
138
|
+
@patch("concurrent.futures.ThreadPoolExecutor", _ImmediateExecutor)
|
|
139
|
+
@patch("fitz.open")
|
|
140
|
+
def test_document_ocr_preserves_default_pixmap_rendering(
|
|
141
|
+
self,
|
|
142
|
+
mock_fitz_open,
|
|
143
|
+
_mock_as_completed,
|
|
144
|
+
mock_client_cls,
|
|
145
|
+
):
|
|
146
|
+
page = _FakePage()
|
|
147
|
+
mock_fitz_open.return_value = _FakePdf([page])
|
|
148
|
+
mock_client_cls.return_value = _FakeClient()
|
|
149
|
+
|
|
150
|
+
converter = DocumentOCRToTextConverter()
|
|
151
|
+
converter.get_document_ocr("dummy.pdf")
|
|
152
|
+
|
|
153
|
+
self.assertEqual(page.pixmap_calls, [{}])
|
|
154
|
+
|
|
115
155
|
def test_default_ocr_max_output_tokens_is_8192(self):
|
|
116
156
|
converter = OCRToTextConverter()
|
|
117
157
|
self.assertEqual(converter.max_output_tokens, 8192)
|
|
@@ -138,6 +138,23 @@ class TestOCRImageDescriptions(unittest.TestCase):
|
|
|
138
138
|
self.assertIsInstance(document_ocr_loader, DocumentOCRLoader)
|
|
139
139
|
self.assertFalse(document_ocr_loader.allow_partial_ocr_failures)
|
|
140
140
|
|
|
141
|
+
def test_forced_document_ocr_passes_custom_render_dpi(self):
|
|
142
|
+
loader = BaseLoader(
|
|
143
|
+
source="local",
|
|
144
|
+
force_ocr=True,
|
|
145
|
+
ocr_render_dpi=200,
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
document_ocr_loader = loader.init_loader_class(
|
|
149
|
+
input="/tmp/example.pdf",
|
|
150
|
+
storage_client={},
|
|
151
|
+
llm_api_key=None,
|
|
152
|
+
source="local",
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
self.assertIsInstance(document_ocr_loader, DocumentOCRLoader)
|
|
156
|
+
self.assertEqual(document_ocr_loader.ocr_render_dpi, 200)
|
|
157
|
+
|
|
141
158
|
def test_google_ocr_converter_builds_augmented_markdown_prompt(self):
|
|
142
159
|
converter = OCRToTextConverter(include_image_descriptions=True)
|
|
143
160
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|