polytext 0.2.8__tar.gz → 0.2.8b2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {polytext-0.2.8 → polytext-0.2.8b2}/PKG-INFO +1 -1
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/audio_to_text.py +2 -2
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/beautiful_text.py +2 -2
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/document_ocr_to_text.py +45 -6
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/ocr_to_text.py +3 -2
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/base.py +43 -7
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/document_ocr.py +7 -0
- polytext-0.2.8b2/polytext/prompts/beautiful_text.py +61 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/prompts/ocr.py +9 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext.egg-info/PKG-INFO +1 -1
- {polytext-0.2.8 → polytext-0.2.8b2}/setup.py +1 -1
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_audio_transcription_model_migration.py +1 -1
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_base_loader_error_mapping.py +47 -0
- polytext-0.2.8/polytext/prompts/beautiful_text.py +0 -43
- {polytext-0.2.8 → polytext-0.2.8b2}/LICENSE +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/README.md +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/__init__.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/__init__.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/base.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/document_ocr_to_text_azure_oai.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/gemini_quality_guards.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/html_to_md.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/md_to_text.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/ocr_to_text_azure_oai.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/pdf.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/text_to_md.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/video_to_audio.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/exceptions/__init__.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/exceptions/base.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/generator/__init__.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/generator/pdf.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/__init__.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/audio.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/aws_auth.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/document.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/downloader/__init__.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/downloader/downloader.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/html.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/markdown.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/notebook.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/ocr.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/plain_text.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/video.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/xml_xbrl.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/youtube.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/youtube_llm.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/processor/__init__.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/processor/audio_chunker.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/processor/text_merger.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/processor/transcript_chunker.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/prompts/__init__.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/prompts/text_merging.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/prompts/text_to_md.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/prompts/transcription.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/utils/__init__.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext/utils/utils.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext.egg-info/SOURCES.txt +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext.egg-info/dependency_links.txt +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext.egg-info/not-zip-safe +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext.egg-info/requires.txt +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/polytext.egg-info/top_level.txt +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/pyproject.toml +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/setup.cfg +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_audio_chunker.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_audio_comparison_helpers.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_aws_auth.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_beautiful_text_manual.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_compare_audio_models.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_compare_document_ocr_to_text_models.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_compare_ocr_to_text_models.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_compare_youtube_models.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_dowload_audio_from_youtube.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_dowload_audio_from_youtube_helpers.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_extracted_text_whitespace.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_gemini_quality_guards.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_audio_transcript_from_gcs.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_customized_pdf_from_markdown.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_document_ocr.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_document_ocr_azure_oai.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_document_text.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_document_text_from_gcs.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_ocr_from_image.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_text_from_markdown.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_video_transcript_from_gcs.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_library.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_markdown_loader_gzip.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_markitdown_html.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_notebook_loader.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_ocr_fallbacks.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_ocr_image_descriptions.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_pain_text.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_pdf_conversion_error.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_python_version_metadata.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_split_audio_with_llm.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_transcribe_s3_images_from_csv.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_transcribe_s3_images_from_csv_script.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_xml_xbrl_loader.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_youtube_gemini_minimal_check.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_youtube_llm_fallbacks.py +0 -0
- {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_youtube_transcript.py +0 -0
|
@@ -93,10 +93,10 @@ def add_line_break_after_each_sentence(text: str) -> str:
|
|
|
93
93
|
continue
|
|
94
94
|
|
|
95
95
|
normalized_line = re.sub(r"\s+", " ", stripped_line)
|
|
96
|
-
normalized_line = re.sub(r"([.!?])\s+", r"\1
|
|
96
|
+
normalized_line = re.sub(r"([.!?])\s+", r"\1\\n ", normalized_line)
|
|
97
97
|
formatted_lines.append(normalized_line)
|
|
98
98
|
|
|
99
|
-
return "
|
|
99
|
+
return "\\n ".join(formatted_lines).strip()
|
|
100
100
|
|
|
101
101
|
|
|
102
102
|
def create_ascii_safe_upload_copy(audio_file: str) -> tuple[str, str | None]:
|
|
@@ -156,7 +156,7 @@ class BeautifulTextConverter:
|
|
|
156
156
|
finalize_nodes()
|
|
157
157
|
return chapters
|
|
158
158
|
|
|
159
|
-
def convert(self, raw_text: str, save_transcript_chunks: bool = False, active_chapters: bool =
|
|
159
|
+
def convert(self, raw_text: str, save_transcript_chunks: bool = False, active_chapters: bool = True) -> dict:
|
|
160
160
|
cleaned_input = (raw_text or "").strip()
|
|
161
161
|
if not cleaned_input:
|
|
162
162
|
result = {
|
|
@@ -204,6 +204,6 @@ class BeautifulTextConverter:
|
|
|
204
204
|
"text_chunks": cleaned_chunks if save_transcript_chunks else "not provided",
|
|
205
205
|
}
|
|
206
206
|
if active_chapters:
|
|
207
|
-
result["markdown_json"] = self._convert_markdown_to_json(final_text)
|
|
207
|
+
# result["markdown_json"] = self._convert_markdown_to_json(final_text)
|
|
208
208
|
result["chapters"] = self._build_chapters(final_text)
|
|
209
209
|
return result
|
|
@@ -13,6 +13,7 @@ from google.api_core import exceptions as google_exceptions
|
|
|
13
13
|
from ..prompts.ocr import (
|
|
14
14
|
OCR_TO_MARKDOWN_NON_LITERAL_FALLBACK_PROMPT,
|
|
15
15
|
OCR_TO_MARKDOWN_PROMPT,
|
|
16
|
+
OCR_TO_PLAIN_TEXT_NON_LITERAL_FALLBACK_PROMPT,
|
|
16
17
|
OCR_TO_PLAIN_TEXT_PROMPT,
|
|
17
18
|
build_ocr_prompt,
|
|
18
19
|
)
|
|
@@ -34,7 +35,7 @@ OCR_TAIL_REPETITION_THRESHOLD = float(os.getenv("OCR_TAIL_REPETITION_THRESHOLD",
|
|
|
34
35
|
OCR_FALLBACK_SOURCE_PATTERN = os.getenv("OCR_FALLBACK_SOURCE_PATTERN", "flash-lite-preview")
|
|
35
36
|
OCR_FALLBACK_MODEL = os.getenv("OCR_FALLBACK_MODEL", "gemini-3-flash-preview")
|
|
36
37
|
OCR_FALLBACK_TEMPERATURE = float(os.getenv("OCR_FALLBACK_TEMPERATURE", "1.0"))
|
|
37
|
-
OCR_FINAL_FALLBACK_MODEL = os.getenv("OCR_FINAL_FALLBACK_MODEL", "gemini-
|
|
38
|
+
OCR_FINAL_FALLBACK_MODEL = os.getenv("OCR_FINAL_FALLBACK_MODEL", "gemini-3.5-flash")
|
|
38
39
|
OCR_PROMPT_VARIANT_DEFAULT = "default"
|
|
39
40
|
OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK = "non_literal_fallback"
|
|
40
41
|
OCR_RETRIABLE_OUTPUT_ERROR_CODES = (996, 997, 999)
|
|
@@ -113,6 +114,7 @@ def get_document_ocr(
|
|
|
113
114
|
ocr_model: str | None = None,
|
|
114
115
|
max_output_tokens: int | None = None,
|
|
115
116
|
include_image_descriptions: bool = False,
|
|
117
|
+
allow_partial_ocr_failures: bool = False,
|
|
116
118
|
):
|
|
117
119
|
"""
|
|
118
120
|
Convenience function to extract text from an image file using OCR, optionally formatted as Markdown.
|
|
@@ -136,6 +138,9 @@ def get_document_ocr(
|
|
|
136
138
|
include_image_descriptions (bool, optional): If True, OCR prompts include
|
|
137
139
|
brief functional descriptions for meaningful non-text images.
|
|
138
140
|
Defaults to False.
|
|
141
|
+
allow_partial_ocr_failures (bool, optional): If True, pages that still
|
|
142
|
+
fail OCR after all retries are recorded inline instead of aborting
|
|
143
|
+
the whole document extraction. Defaults to False.
|
|
139
144
|
|
|
140
145
|
Returns:
|
|
141
146
|
dict: Dictionary containing the OCR results and metadata.
|
|
@@ -149,6 +154,7 @@ def get_document_ocr(
|
|
|
149
154
|
timeout_minutes=timeout_minutes,
|
|
150
155
|
max_output_tokens=max_output_tokens,
|
|
151
156
|
include_image_descriptions=include_image_descriptions,
|
|
157
|
+
allow_partial_ocr_failures=allow_partial_ocr_failures,
|
|
152
158
|
)
|
|
153
159
|
return converter.get_document_ocr(document_for_ocr)
|
|
154
160
|
|
|
@@ -157,7 +163,8 @@ class DocumentOCRToTextConverter:
|
|
|
157
163
|
markdown_output=True, llm_api_key=None, target_size=1, temp_dir="temp",
|
|
158
164
|
page_range=None, timeout_minutes: int = None, fallback_stage: int = 0,
|
|
159
165
|
max_output_tokens: int | None = None, include_image_descriptions: bool = False,
|
|
160
|
-
prompt_variant: str = OCR_PROMPT_VARIANT_DEFAULT
|
|
166
|
+
prompt_variant: str = OCR_PROMPT_VARIANT_DEFAULT,
|
|
167
|
+
allow_partial_ocr_failures: bool = False):
|
|
161
168
|
"""
|
|
162
169
|
Initialize the DocumentOCRToTextConverter class with specified OCR model and formatting options.
|
|
163
170
|
|
|
@@ -182,6 +189,9 @@ class DocumentOCRToTextConverter:
|
|
|
182
189
|
Defaults to False.
|
|
183
190
|
prompt_variant (str, optional): Prompt variant used by this attempt.
|
|
184
191
|
Defaults to "default".
|
|
192
|
+
allow_partial_ocr_failures (bool, optional): If True, pages that still
|
|
193
|
+
fail OCR after all retries are recorded inline instead of aborting
|
|
194
|
+
the whole document extraction. Defaults to False.
|
|
185
195
|
|
|
186
196
|
Raises:
|
|
187
197
|
OSError: If temp directory creation fails
|
|
@@ -196,6 +206,7 @@ class DocumentOCRToTextConverter:
|
|
|
196
206
|
self.timeout_minutes = timeout_minutes
|
|
197
207
|
self.include_image_descriptions = include_image_descriptions
|
|
198
208
|
self.prompt_variant = prompt_variant
|
|
209
|
+
self.allow_partial_ocr_failures = allow_partial_ocr_failures
|
|
199
210
|
requested_output_tokens = OCR_MAX_OUTPUT_TOKENS if max_output_tokens is None else max_output_tokens
|
|
200
211
|
self.max_output_tokens = max(requested_output_tokens, OCR_MIN_OUTPUT_TOKENS)
|
|
201
212
|
self.fallback_stage = fallback_stage
|
|
@@ -212,6 +223,8 @@ class DocumentOCRToTextConverter:
|
|
|
212
223
|
def _build_prompt_template(self) -> str:
|
|
213
224
|
if self.markdown_output and self.prompt_variant == OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
|
|
214
225
|
base_prompt = OCR_TO_MARKDOWN_NON_LITERAL_FALLBACK_PROMPT
|
|
226
|
+
elif not self.markdown_output and self.prompt_variant == OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
|
|
227
|
+
base_prompt = OCR_TO_PLAIN_TEXT_NON_LITERAL_FALLBACK_PROMPT
|
|
215
228
|
elif self.markdown_output:
|
|
216
229
|
base_prompt = OCR_TO_MARKDOWN_PROMPT
|
|
217
230
|
else:
|
|
@@ -224,8 +237,6 @@ class DocumentOCRToTextConverter:
|
|
|
224
237
|
def should_prompt_fallback_retry(self, error: EmptyDocument) -> bool:
|
|
225
238
|
if self.fallback_stage != 0:
|
|
226
239
|
return False
|
|
227
|
-
if not self.markdown_output:
|
|
228
|
-
return False
|
|
229
240
|
if self.prompt_variant == OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
|
|
230
241
|
return False
|
|
231
242
|
return error.code in OCR_RETRIABLE_OUTPUT_ERROR_CODES
|
|
@@ -283,6 +294,7 @@ class DocumentOCRToTextConverter:
|
|
|
283
294
|
max_output_tokens=self.max_output_tokens,
|
|
284
295
|
include_image_descriptions=self.include_image_descriptions,
|
|
285
296
|
prompt_variant=resolved_prompt_variant,
|
|
297
|
+
allow_partial_ocr_failures=self.allow_partial_ocr_failures,
|
|
286
298
|
)
|
|
287
299
|
result = fallback_converter.get_ocr(
|
|
288
300
|
file_for_ocr=file_for_ocr,
|
|
@@ -541,7 +553,26 @@ class DocumentOCRToTextConverter:
|
|
|
541
553
|
pix.save(temp_image_path)
|
|
542
554
|
|
|
543
555
|
# Perform OCR on the page
|
|
544
|
-
|
|
556
|
+
try:
|
|
557
|
+
ocr_result = self.get_ocr(temp_image_path)
|
|
558
|
+
except EmptyDocument as error:
|
|
559
|
+
if not self.allow_partial_ocr_failures:
|
|
560
|
+
raise
|
|
561
|
+
logger.warning(
|
|
562
|
+
"Document OCR failed on page %s after retries; keeping partial document because allow_partial_ocr_failures=True: %s",
|
|
563
|
+
page_num + 1,
|
|
564
|
+
error.message,
|
|
565
|
+
)
|
|
566
|
+
ocr_result = {
|
|
567
|
+
"text": "",
|
|
568
|
+
"completion_tokens": 0,
|
|
569
|
+
"prompt_tokens": 0,
|
|
570
|
+
"completion_model": self.ocr_model,
|
|
571
|
+
"completion_model_provider": self.ocr_model_provider,
|
|
572
|
+
"text_chunks": "not provided",
|
|
573
|
+
"page_error": True,
|
|
574
|
+
"page_error_reason": error.message,
|
|
575
|
+
}
|
|
545
576
|
return page_num, ocr_result
|
|
546
577
|
|
|
547
578
|
finally:
|
|
@@ -574,11 +605,17 @@ class DocumentOCRToTextConverter:
|
|
|
574
605
|
all_text = []
|
|
575
606
|
total_completion_tokens = 0
|
|
576
607
|
total_prompt_tokens = 0
|
|
608
|
+
failed_pages = []
|
|
577
609
|
|
|
578
|
-
for
|
|
610
|
+
for page_num, ocr_result in results:
|
|
579
611
|
all_text.append(f"{ocr_result['text']}\n")
|
|
580
612
|
total_completion_tokens += ocr_result['completion_tokens']
|
|
581
613
|
total_prompt_tokens += ocr_result['prompt_tokens']
|
|
614
|
+
if ocr_result.get("page_error"):
|
|
615
|
+
failed_pages.append({
|
|
616
|
+
"page": page_num + 1,
|
|
617
|
+
"reason": ocr_result.get("page_error_reason", "unknown"),
|
|
618
|
+
})
|
|
582
619
|
|
|
583
620
|
pdf.close()
|
|
584
621
|
|
|
@@ -589,6 +626,8 @@ class DocumentOCRToTextConverter:
|
|
|
589
626
|
"completion_model": self.ocr_model,
|
|
590
627
|
"completion_model_provider": self.ocr_model_provider,
|
|
591
628
|
"text_chunks": "not provided",
|
|
629
|
+
"ocr_failed_pages": [item["page"] for item in failed_pages],
|
|
630
|
+
"ocr_failed_pages_detail": failed_pages,
|
|
592
631
|
}
|
|
593
632
|
|
|
594
633
|
return final_result_dict
|
|
@@ -13,6 +13,7 @@ from google.api_core import exceptions as google_exceptions
|
|
|
13
13
|
from ..prompts.ocr import (
|
|
14
14
|
OCR_TO_MARKDOWN_NON_LITERAL_FALLBACK_PROMPT,
|
|
15
15
|
OCR_TO_MARKDOWN_PROMPT,
|
|
16
|
+
OCR_TO_PLAIN_TEXT_NON_LITERAL_FALLBACK_PROMPT,
|
|
16
17
|
OCR_TO_PLAIN_TEXT_PROMPT,
|
|
17
18
|
build_ocr_prompt,
|
|
18
19
|
)
|
|
@@ -207,6 +208,8 @@ class OCRToTextConverter:
|
|
|
207
208
|
def _build_prompt_template(self) -> str:
|
|
208
209
|
if self.markdown_output and self.prompt_variant == OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
|
|
209
210
|
base_prompt = OCR_TO_MARKDOWN_NON_LITERAL_FALLBACK_PROMPT
|
|
211
|
+
elif not self.markdown_output and self.prompt_variant == OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
|
|
212
|
+
base_prompt = OCR_TO_PLAIN_TEXT_NON_LITERAL_FALLBACK_PROMPT
|
|
210
213
|
elif self.markdown_output:
|
|
211
214
|
base_prompt = OCR_TO_MARKDOWN_PROMPT
|
|
212
215
|
else:
|
|
@@ -219,8 +222,6 @@ class OCRToTextConverter:
|
|
|
219
222
|
def should_prompt_fallback_retry(self, error: EmptyDocument) -> bool:
|
|
220
223
|
if self.fallback_stage != 0:
|
|
221
224
|
return False
|
|
222
|
-
if not self.markdown_output:
|
|
223
|
-
return False
|
|
224
225
|
if self.prompt_variant == OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
|
|
225
226
|
return False
|
|
226
227
|
return error.code in OCR_RETRIABLE_OUTPUT_ERROR_CODES
|
|
@@ -90,7 +90,9 @@ def _raise_empty_document_loader_error(error: EmptyDocument) -> None:
|
|
|
90
90
|
class BaseLoader:
|
|
91
91
|
def __init__(self, markdown_output=True, llm_api_key=None, provider: str = "google", temp_dir: str = "temp",
|
|
92
92
|
ocr_model: str = "gpt-5-mini", timeout_minutes: int | None = None,
|
|
93
|
-
include_image_descriptions: bool | None = None,
|
|
93
|
+
include_image_descriptions: bool | None = None,
|
|
94
|
+
force_ocr: bool = False,
|
|
95
|
+
**kwargs):
|
|
94
96
|
"""
|
|
95
97
|
Initialize the BaseLoader with cloud storage and LLM configurations.
|
|
96
98
|
|
|
@@ -108,6 +110,8 @@ class BaseLoader:
|
|
|
108
110
|
include_image_descriptions (bool | None, optional): If True, OCR prompts
|
|
109
111
|
include brief functional descriptions for meaningful non-text images.
|
|
110
112
|
If None, defaults from OCR_INCLUDE_IMAGE_DESCRIPTIONS. Defaults to None.
|
|
113
|
+
force_ocr (bool, optional): If True, supported document files are routed to
|
|
114
|
+
OCRLoader instead of the standard DocumentLoader. Defaults to False.
|
|
111
115
|
**kwargs: Additional keyword arguments to pass to the underlying loader or extraction logic.
|
|
112
116
|
- target_size (int, optional): Target file size in bytes. Defaults to 1MB
|
|
113
117
|
- source (str): Source of the document. Must be either "cloud" or "local"
|
|
@@ -131,6 +135,7 @@ class BaseLoader:
|
|
|
131
135
|
if include_image_descriptions is None
|
|
132
136
|
else include_image_descriptions
|
|
133
137
|
)
|
|
138
|
+
self.force_ocr = force_ocr
|
|
134
139
|
self.kwargs = kwargs
|
|
135
140
|
self.target_size = kwargs.get("target_size", 1)
|
|
136
141
|
self.source = kwargs.get("source", "cloud")
|
|
@@ -259,8 +264,14 @@ class BaseLoader:
|
|
|
259
264
|
cleanup_result = converter.convert(
|
|
260
265
|
raw_text=raw_result["text"],
|
|
261
266
|
save_transcript_chunks=kwargs.get("save_transcript_chunks", self.save_transcript_chunks),
|
|
262
|
-
active_chapters=kwargs.get("active_chapters",
|
|
267
|
+
active_chapters=kwargs.get("active_chapters", True),
|
|
263
268
|
)
|
|
269
|
+
if not cleanup_result.get("chapters"):
|
|
270
|
+
raise LoaderError(
|
|
271
|
+
message="No chapters detected",
|
|
272
|
+
status=422,
|
|
273
|
+
code="NO_CHAPTERS_DETECTED",
|
|
274
|
+
)
|
|
264
275
|
|
|
265
276
|
total_completion_tokens = raw_result.get("completion_tokens", 0) + cleanup_result.get("completion_tokens", 0)
|
|
266
277
|
total_prompt_tokens = raw_result.get("prompt_tokens", 0) + cleanup_result.get("prompt_tokens", 0)
|
|
@@ -405,8 +416,7 @@ class BaseLoader:
|
|
|
405
416
|
if path_without_query:
|
|
406
417
|
_, file_extension = os.path.splitext(path_without_query)
|
|
407
418
|
else: # If is local file path (without schema)
|
|
408
|
-
|
|
409
|
-
_, file_extension = os.path.splitext(input)
|
|
419
|
+
_, file_extension = os.path.splitext(input)
|
|
410
420
|
|
|
411
421
|
if file_extension:
|
|
412
422
|
file_extension = file_extension.lower()
|
|
@@ -436,7 +446,29 @@ class BaseLoader:
|
|
|
436
446
|
)
|
|
437
447
|
elif mime_type:
|
|
438
448
|
if file_extension in [".pdf", ".xlsx", ".docx", ".txt", ".csv", ".odt", ".pptx", ".xls", ".doc", ".ppt", ".rtf"]:
|
|
439
|
-
|
|
449
|
+
document_kwargs = {k: v for k, v in kwargs.items() if k != "source"}
|
|
450
|
+
if self.force_ocr:
|
|
451
|
+
|
|
452
|
+
return DocumentOCRLoader(
|
|
453
|
+
source=self.source,
|
|
454
|
+
llm_api_key=llm_api_key,
|
|
455
|
+
markdown_output=self.markdown_output,
|
|
456
|
+
temp_dir=self.temp_dir,
|
|
457
|
+
timeout_minutes=self.timeout_minutes,
|
|
458
|
+
ocr_provider=self.provider,
|
|
459
|
+
ocr_model=self.ocr_model,
|
|
460
|
+
include_image_descriptions=self.include_image_descriptions,
|
|
461
|
+
allow_partial_ocr_failures=True,
|
|
462
|
+
**document_kwargs,
|
|
463
|
+
)
|
|
464
|
+
|
|
465
|
+
return DocumentLoader(
|
|
466
|
+
source=self.source,
|
|
467
|
+
markdown_output=self.markdown_output,
|
|
468
|
+
temp_dir=self.temp_dir,
|
|
469
|
+
timeout_minutes=self.timeout_minutes,
|
|
470
|
+
**document_kwargs,
|
|
471
|
+
)
|
|
440
472
|
elif mime_type.startswith("audio/"):
|
|
441
473
|
audio_kwargs = {**kwargs, "is_output_audio_raw": self.is_output_audio_raw}
|
|
442
474
|
return AudioLoader(llm_api_key=llm_api_key, markdown_output=self.markdown_output, temp_dir=self.temp_dir, timeout_minutes=self.timeout_minutes, **audio_kwargs)
|
|
@@ -565,14 +597,18 @@ class BaseLoader:
|
|
|
565
597
|
|
|
566
598
|
result_dict["text"] = clean_extracted_text_whitespace(remove_markdown_strip(result_dict["text"]))
|
|
567
599
|
|
|
568
|
-
|
|
600
|
+
final_result = {
|
|
569
601
|
"text": result_dict["text"],
|
|
570
602
|
"completion_tokens": result_dict["completion_tokens"],
|
|
571
603
|
"prompt_tokens": result_dict["prompt_tokens"],
|
|
572
604
|
"output_list": [result_dict],
|
|
573
605
|
}
|
|
574
606
|
|
|
575
|
-
|
|
607
|
+
for metadata_key in ("ocr_failed_pages", "ocr_failed_pages_detail"):
|
|
608
|
+
if metadata_key in result_dict:
|
|
609
|
+
final_result[metadata_key] = result_dict[metadata_key]
|
|
610
|
+
|
|
611
|
+
return final_result
|
|
576
612
|
|
|
577
613
|
# Helper methods
|
|
578
614
|
@staticmethod
|
|
@@ -46,6 +46,7 @@ class DocumentOCRLoader:
|
|
|
46
46
|
ocr_provider: str = "google",
|
|
47
47
|
ocr_model: str | None = None,
|
|
48
48
|
include_image_descriptions: bool = False,
|
|
49
|
+
allow_partial_ocr_failures: bool = False,
|
|
49
50
|
**kwargs
|
|
50
51
|
):
|
|
51
52
|
"""
|
|
@@ -82,6 +83,9 @@ class DocumentOCRLoader:
|
|
|
82
83
|
include_image_descriptions (bool, optional): If True, OCR prompts include
|
|
83
84
|
brief functional descriptions for meaningful non-text images.
|
|
84
85
|
Defaults to False.
|
|
86
|
+
allow_partial_ocr_failures (bool, optional): If True, pages that still
|
|
87
|
+
fail OCR after all retries are recorded inline instead of aborting
|
|
88
|
+
the whole document extraction. Defaults to False.
|
|
85
89
|
**kwargs:
|
|
86
90
|
max_output_tokens (int, optional): Maximum Gemini output tokens for
|
|
87
91
|
Google document OCR generation.
|
|
@@ -105,6 +109,7 @@ class DocumentOCRLoader:
|
|
|
105
109
|
self.ocr_provider = (ocr_provider or "google").lower()
|
|
106
110
|
self.ocr_model = ocr_model
|
|
107
111
|
self.include_image_descriptions = include_image_descriptions
|
|
112
|
+
self.allow_partial_ocr_failures = allow_partial_ocr_failures
|
|
108
113
|
self.max_output_tokens = kwargs.get("max_output_tokens")
|
|
109
114
|
|
|
110
115
|
# Set up custom temp directory
|
|
@@ -254,6 +259,7 @@ class DocumentOCRLoader:
|
|
|
254
259
|
timeout_minutes=self.timeout_minutes,
|
|
255
260
|
ocr_model=self.ocr_model or None,
|
|
256
261
|
include_image_descriptions=self.include_image_descriptions,
|
|
262
|
+
allow_partial_ocr_failures=self.allow_partial_ocr_failures,
|
|
257
263
|
)
|
|
258
264
|
else:
|
|
259
265
|
result_dict = ocr_fn(
|
|
@@ -270,6 +276,7 @@ class DocumentOCRLoader:
|
|
|
270
276
|
),
|
|
271
277
|
max_output_tokens=self.max_output_tokens,
|
|
272
278
|
include_image_descriptions=self.include_image_descriptions,
|
|
279
|
+
allow_partial_ocr_failures=self.allow_partial_ocr_failures,
|
|
273
280
|
)
|
|
274
281
|
|
|
275
282
|
result_dict["type"] = self.type
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
BEAUTIFUL_TEXT_PROMPT = """
|
|
2
|
+
You are cleaning a spoken transcript into faithful Markdown.
|
|
3
|
+
|
|
4
|
+
This is NOT summarization.
|
|
5
|
+
This is NOT rewriting.
|
|
6
|
+
This is NOT editorial adaptation.
|
|
7
|
+
This is a cleaned transcript.
|
|
8
|
+
|
|
9
|
+
Your task is to remove only clearly meaningless noise while preserving the original spoken wording,
|
|
10
|
+
sequence, tone, rhythm, and reasoning as closely as possible.
|
|
11
|
+
|
|
12
|
+
REMOVE ONLY:
|
|
13
|
+
- obvious filler sounds such as "eh", "uhm", "mmh" when they clearly add no meaning
|
|
14
|
+
- accidental duplicated words such as "di di", "che che", "da da"
|
|
15
|
+
- clearly aborted false starts that add no meaning
|
|
16
|
+
- irrelevant overlap fragments between speakers
|
|
17
|
+
|
|
18
|
+
IF THERE IS ANY DOUBT, KEEP THE ORIGINAL WORDING.
|
|
19
|
+
|
|
20
|
+
PRESERVE STRICTLY:
|
|
21
|
+
- the original wording
|
|
22
|
+
- the original order of ideas
|
|
23
|
+
- the original paragraph flow
|
|
24
|
+
- the original tone and conversational style
|
|
25
|
+
- repetitions that still carry emphasis, rhythm, hesitation, or meaning
|
|
26
|
+
- colloquial phrasing and spoken transitions when meaningful
|
|
27
|
+
|
|
28
|
+
DO NOT:
|
|
29
|
+
- summarize
|
|
30
|
+
- compress
|
|
31
|
+
- simplify
|
|
32
|
+
- polish into formal written prose
|
|
33
|
+
- merge multiple spoken sentences into a shorter reformulation
|
|
34
|
+
- turn the transcript into an article, essay, report, or explanatory text
|
|
35
|
+
- replace words with better synonyms
|
|
36
|
+
- add explanations, transitions, or inferred content
|
|
37
|
+
- add interpretive conclusions
|
|
38
|
+
|
|
39
|
+
FORMATTING:
|
|
40
|
+
- output Markdown only
|
|
41
|
+
- preserve the transcript as a cleaned spoken transcript, not as a rewritten article
|
|
42
|
+
- use paragraphs, but do not heavily reorganize the flow
|
|
43
|
+
- headings may be generated editorially, but only to label the topic of the following block
|
|
44
|
+
- headings must be short, neutral, and strictly supported by the text below
|
|
45
|
+
- the final Markdown must contain at least one heading
|
|
46
|
+
- if the text is short or has weak structure, add one minimal heading only
|
|
47
|
+
- do not convert prose into bullet lists unless the speaker is explicitly enumerating points
|
|
48
|
+
- use emphasis sparingly
|
|
49
|
+
- you may add light Markdown emphasis to improve scanability, but only locally and without rewriting the sentence
|
|
50
|
+
- use **bold** for clearly salient entities already present in the source, such as names, products, platforms, institutions, laws, or central concepts
|
|
51
|
+
- use *italics* sparingly for contextual labels, foreign expressions, or technical terms when this remains clearly faithful to the source
|
|
52
|
+
- do not apply emphasis to large portions of text
|
|
53
|
+
- do not use emphasis as a substitute for rewriting, summarizing, or restructuring
|
|
54
|
+
- do not add emphasis just to make the text nicer
|
|
55
|
+
- do not add code fences or commentary
|
|
56
|
+
|
|
57
|
+
FINAL CHECK:
|
|
58
|
+
- every output sentence must remain closely traceable to the input
|
|
59
|
+
- prefer awkward fidelity over elegant rewriting
|
|
60
|
+
- headings may be editorially generated, but body text must remain maximally faithful
|
|
61
|
+
"""
|
|
@@ -28,6 +28,15 @@ Maintain paragraph breaks and formatting.
|
|
|
28
28
|
Your output must be a plain text.
|
|
29
29
|
"""
|
|
30
30
|
|
|
31
|
+
OCR_TO_PLAIN_TEXT_NON_LITERAL_FALLBACK_PROMPT = """
|
|
32
|
+
Rephrase and reorganize the text content in a coherent plain text structure.
|
|
33
|
+
Do not transcribe the text verbatim, but preserve the meaning of the original content without omitting anything, even apparently minor details.
|
|
34
|
+
Pay special attention to tables, columns, headers, and any structured content.
|
|
35
|
+
Maintain paragraph breaks and formatting.
|
|
36
|
+
Your output must be a plain text.
|
|
37
|
+
In case no readable text is present, write exactly "no readable text present".
|
|
38
|
+
"""
|
|
39
|
+
|
|
31
40
|
OCR_IMAGE_DESCRIPTION_INSTRUCTIONS = """
|
|
32
41
|
Image description instructions:
|
|
33
42
|
|
|
@@ -51,7 +51,7 @@ def get_requirements(*requirements_file):
|
|
|
51
51
|
|
|
52
52
|
setup(
|
|
53
53
|
name='polytext',
|
|
54
|
-
version='0.2.
|
|
54
|
+
version='0.2.8b2',
|
|
55
55
|
url='https://github.com/docsity/polytext',
|
|
56
56
|
# download_url='https://github.com/pualien/py-polytext/archive/0.1.23.tar.gz',
|
|
57
57
|
license='MIT',
|
|
@@ -137,7 +137,7 @@ class TestAudioTranscriptionModelMigration(unittest.TestCase):
|
|
|
137
137
|
|
|
138
138
|
self.assertEqual(
|
|
139
139
|
formatted,
|
|
140
|
-
"Prima frase
|
|
140
|
+
"Prima frase.\\n Seconda frase?\\n Terza frase!\\n ## Titolo\\n Quarta frase.\\n Quinta frase.",
|
|
141
141
|
)
|
|
142
142
|
|
|
143
143
|
def test_normalize_no_human_speech_marker_returns_empty_for_marker_only(self):
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import unittest
|
|
2
|
+
from types import SimpleNamespace
|
|
2
3
|
from unittest.mock import Mock, patch
|
|
3
4
|
|
|
4
5
|
from polytext.exceptions import ConversionError, EmptyDocument, LoaderError
|
|
@@ -40,6 +41,24 @@ class _FallbackFailingBaseLoader(BaseLoader):
|
|
|
40
41
|
return _FailingLoader(self.initial_error)
|
|
41
42
|
|
|
42
43
|
|
|
44
|
+
class _BeautifulTextLoader(BaseLoader):
|
|
45
|
+
def __init__(self, raw_result=None, **kwargs):
|
|
46
|
+
super().__init__(**kwargs)
|
|
47
|
+
self.raw_result = raw_result or {
|
|
48
|
+
"text": "raw text",
|
|
49
|
+
"completion_tokens": 0,
|
|
50
|
+
"prompt_tokens": 0,
|
|
51
|
+
"completion_model": "not provided",
|
|
52
|
+
"completion_model_provider": "not provided",
|
|
53
|
+
"text_chunks": "not provided",
|
|
54
|
+
"type": "text",
|
|
55
|
+
"input": "dummy input",
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
def extract_raw_text_for_beautiful_text(self, input_value: str, **kwargs) -> dict:
|
|
59
|
+
return self.raw_result
|
|
60
|
+
|
|
61
|
+
|
|
43
62
|
class TestBaseLoaderErrorMapping(unittest.TestCase):
|
|
44
63
|
def test_llm_output_empty_document_codes_are_raised_as_loader_errors(self):
|
|
45
64
|
cases = [
|
|
@@ -143,6 +162,34 @@ class TestBaseLoaderErrorMapping(unittest.TestCase):
|
|
|
143
162
|
mock_exception.assert_not_called()
|
|
144
163
|
sentry_sdk.capture_exception.assert_called_once_with(conversion_error)
|
|
145
164
|
|
|
165
|
+
def test_beautiful_text_without_detected_chapters_is_raised_as_loader_error(self):
|
|
166
|
+
loader = _BeautifulTextLoader()
|
|
167
|
+
cleanup_result = {
|
|
168
|
+
"text": "Short text without headings.",
|
|
169
|
+
"completion_tokens": 1,
|
|
170
|
+
"prompt_tokens": 1,
|
|
171
|
+
"completion_model": "gemini-3.1-flash-lite",
|
|
172
|
+
"completion_model_provider": "google",
|
|
173
|
+
"text_chunks": "not provided",
|
|
174
|
+
"markdown_json": {"root": ["Short text without headings."]},
|
|
175
|
+
"chapters": [],
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
fake_converter_class = Mock()
|
|
179
|
+
fake_converter_class.return_value.convert.return_value = cleanup_result
|
|
180
|
+
|
|
181
|
+
with patch.dict(
|
|
182
|
+
"sys.modules",
|
|
183
|
+
{"polytext.converter.beautiful_text": SimpleNamespace(BeautifulTextConverter=fake_converter_class)},
|
|
184
|
+
):
|
|
185
|
+
with self.assertRaises(LoaderError) as error_context:
|
|
186
|
+
loader.get_beautiful_text(["dummy input"], active_chapters=True)
|
|
187
|
+
|
|
188
|
+
error = error_context.exception
|
|
189
|
+
self.assertEqual(error.status, 422)
|
|
190
|
+
self.assertEqual(error.code, "NO_CHAPTERS_DETECTED")
|
|
191
|
+
self.assertEqual(error.message, "No chapters detected")
|
|
192
|
+
|
|
146
193
|
|
|
147
194
|
if __name__ == "__main__":
|
|
148
195
|
unittest.main()
|
|
@@ -1,43 +0,0 @@
|
|
|
1
|
-
BEAUTIFUL_TEXT_PROMPT = """
|
|
2
|
-
You are an editor specialized in cleaning spoken transcripts and raw text into faithful Markdown.
|
|
3
|
-
This is not summarization. This is not rewriting. This is a cleaned transcript or cleaned source text.
|
|
4
|
-
|
|
5
|
-
Your task is to remove only accidental noise while preserving the speaker's or author's original words,
|
|
6
|
-
phrasing, reasoning, tone, and sequence of ideas as faithfully as possible.
|
|
7
|
-
|
|
8
|
-
REMOVE ONLY:
|
|
9
|
-
- non-meaningful fillers such as "eh", "uhm", "diciamo", "eccetera eccetera", "no?" when used only as filler
|
|
10
|
-
- redundant "quindi", "appunto", "comunque" when they are only conversational padding
|
|
11
|
-
- accidental repeated words such as "di di", "da da", "che che"
|
|
12
|
-
- false starts and self-corrections only when they do not carry meaning
|
|
13
|
-
- irrelevant overlap fragments between speakers
|
|
14
|
-
|
|
15
|
-
PRESERVE COMPLETELY:
|
|
16
|
-
- the original wording and sentence structure, even if colloquial
|
|
17
|
-
- technical terms and proper nouns exactly
|
|
18
|
-
- the original tone and register
|
|
19
|
-
- reasoning, opinions, nuances, and meaningful uncertainty
|
|
20
|
-
- the logical order of the discussion
|
|
21
|
-
|
|
22
|
-
DO NOT:
|
|
23
|
-
- rewrite sentences in a more elegant style
|
|
24
|
-
- replace words with synonyms
|
|
25
|
-
- summarize, compress, or simplify concepts
|
|
26
|
-
- add explanations, transitions, or missing content
|
|
27
|
-
- correct the speaker's opinions or inaccuracies
|
|
28
|
-
- make the language more formal than the original
|
|
29
|
-
|
|
30
|
-
FORMATTING:
|
|
31
|
-
- output Markdown only
|
|
32
|
-
- use paragraphs to separate thematic blocks
|
|
33
|
-
- add headings only when the speaker explicitly introduces a new topic
|
|
34
|
-
- use bullet lists or numbered lists only when the source explicitly enumerates items or when the sequence is clearly list-shaped
|
|
35
|
-
- use emphasis sparingly and only when grounded in the original text
|
|
36
|
-
- use **bold** for key information and important concepts, and *italics* for subtle emphasis or contextual terms in every chapter and paragraph whenever they improve readability and understanding
|
|
37
|
-
- do not add code fences
|
|
38
|
-
- do not add introductions or commentary
|
|
39
|
-
|
|
40
|
-
FINAL CHECK:
|
|
41
|
-
- every sentence in the output must be traceable to an equivalent sentence in the input
|
|
42
|
-
- if a sentence cannot be grounded in the input, remove it
|
|
43
|
-
"""
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|