polytext 0.2.8__tar.gz → 0.2.8b2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. {polytext-0.2.8 → polytext-0.2.8b2}/PKG-INFO +1 -1
  2. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/audio_to_text.py +2 -2
  3. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/beautiful_text.py +2 -2
  4. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/document_ocr_to_text.py +45 -6
  5. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/ocr_to_text.py +3 -2
  6. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/base.py +43 -7
  7. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/document_ocr.py +7 -0
  8. polytext-0.2.8b2/polytext/prompts/beautiful_text.py +61 -0
  9. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/prompts/ocr.py +9 -0
  10. {polytext-0.2.8 → polytext-0.2.8b2}/polytext.egg-info/PKG-INFO +1 -1
  11. {polytext-0.2.8 → polytext-0.2.8b2}/setup.py +1 -1
  12. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_audio_transcription_model_migration.py +1 -1
  13. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_base_loader_error_mapping.py +47 -0
  14. polytext-0.2.8/polytext/prompts/beautiful_text.py +0 -43
  15. {polytext-0.2.8 → polytext-0.2.8b2}/LICENSE +0 -0
  16. {polytext-0.2.8 → polytext-0.2.8b2}/README.md +0 -0
  17. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/__init__.py +0 -0
  18. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/__init__.py +0 -0
  19. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/base.py +0 -0
  20. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/document_ocr_to_text_azure_oai.py +0 -0
  21. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/gemini_quality_guards.py +0 -0
  22. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/html_to_md.py +0 -0
  23. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/md_to_text.py +0 -0
  24. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/ocr_to_text_azure_oai.py +0 -0
  25. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/pdf.py +0 -0
  26. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/text_to_md.py +0 -0
  27. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/converter/video_to_audio.py +0 -0
  28. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/exceptions/__init__.py +0 -0
  29. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/exceptions/base.py +0 -0
  30. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/generator/__init__.py +0 -0
  31. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/generator/pdf.py +0 -0
  32. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/__init__.py +0 -0
  33. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/audio.py +0 -0
  34. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/aws_auth.py +0 -0
  35. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/document.py +0 -0
  36. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/downloader/__init__.py +0 -0
  37. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/downloader/downloader.py +0 -0
  38. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/html.py +0 -0
  39. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/markdown.py +0 -0
  40. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/notebook.py +0 -0
  41. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/ocr.py +0 -0
  42. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/plain_text.py +0 -0
  43. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/video.py +0 -0
  44. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/xml_xbrl.py +0 -0
  45. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/youtube.py +0 -0
  46. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/loader/youtube_llm.py +0 -0
  47. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/processor/__init__.py +0 -0
  48. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/processor/audio_chunker.py +0 -0
  49. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/processor/text_merger.py +0 -0
  50. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/processor/transcript_chunker.py +0 -0
  51. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/prompts/__init__.py +0 -0
  52. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/prompts/text_merging.py +0 -0
  53. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/prompts/text_to_md.py +0 -0
  54. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/prompts/transcription.py +0 -0
  55. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/utils/__init__.py +0 -0
  56. {polytext-0.2.8 → polytext-0.2.8b2}/polytext/utils/utils.py +0 -0
  57. {polytext-0.2.8 → polytext-0.2.8b2}/polytext.egg-info/SOURCES.txt +0 -0
  58. {polytext-0.2.8 → polytext-0.2.8b2}/polytext.egg-info/dependency_links.txt +0 -0
  59. {polytext-0.2.8 → polytext-0.2.8b2}/polytext.egg-info/not-zip-safe +0 -0
  60. {polytext-0.2.8 → polytext-0.2.8b2}/polytext.egg-info/requires.txt +0 -0
  61. {polytext-0.2.8 → polytext-0.2.8b2}/polytext.egg-info/top_level.txt +0 -0
  62. {polytext-0.2.8 → polytext-0.2.8b2}/pyproject.toml +0 -0
  63. {polytext-0.2.8 → polytext-0.2.8b2}/setup.cfg +0 -0
  64. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_audio_chunker.py +0 -0
  65. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_audio_comparison_helpers.py +0 -0
  66. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_aws_auth.py +0 -0
  67. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_beautiful_text_manual.py +0 -0
  68. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_compare_audio_models.py +0 -0
  69. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_compare_document_ocr_to_text_models.py +0 -0
  70. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_compare_ocr_to_text_models.py +0 -0
  71. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_compare_youtube_models.py +0 -0
  72. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_dowload_audio_from_youtube.py +0 -0
  73. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_dowload_audio_from_youtube_helpers.py +0 -0
  74. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_extracted_text_whitespace.py +0 -0
  75. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_gemini_quality_guards.py +0 -0
  76. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_audio_transcript_from_gcs.py +0 -0
  77. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_customized_pdf_from_markdown.py +0 -0
  78. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_document_ocr.py +0 -0
  79. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_document_ocr_azure_oai.py +0 -0
  80. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_document_text.py +0 -0
  81. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_document_text_from_gcs.py +0 -0
  82. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_ocr_from_image.py +0 -0
  83. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_text_from_markdown.py +0 -0
  84. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_get_video_transcript_from_gcs.py +0 -0
  85. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_library.py +0 -0
  86. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_markdown_loader_gzip.py +0 -0
  87. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_markitdown_html.py +0 -0
  88. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_notebook_loader.py +0 -0
  89. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_ocr_fallbacks.py +0 -0
  90. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_ocr_image_descriptions.py +0 -0
  91. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_pain_text.py +0 -0
  92. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_pdf_conversion_error.py +0 -0
  93. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_python_version_metadata.py +0 -0
  94. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_split_audio_with_llm.py +0 -0
  95. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_transcribe_s3_images_from_csv.py +0 -0
  96. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_transcribe_s3_images_from_csv_script.py +0 -0
  97. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_xml_xbrl_loader.py +0 -0
  98. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_youtube_gemini_minimal_check.py +0 -0
  99. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_youtube_llm_fallbacks.py +0 -0
  100. {polytext-0.2.8 → polytext-0.2.8b2}/tests/test_youtube_transcript.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: polytext
3
- Version: 0.2.8
3
+ Version: 0.2.8b2
4
4
  Summary: Python utilities to simplify document files management
5
5
  Home-page: https://github.com/docsity/polytext
6
6
  Author: Matteo Senardi
@@ -93,10 +93,10 @@ def add_line_break_after_each_sentence(text: str) -> str:
93
93
  continue
94
94
 
95
95
  normalized_line = re.sub(r"\s+", " ", stripped_line)
96
- normalized_line = re.sub(r"([.!?])\s+", r"\1\n", normalized_line)
96
+ normalized_line = re.sub(r"([.!?])\s+", r"\1\\n ", normalized_line)
97
97
  formatted_lines.append(normalized_line)
98
98
 
99
- return "\n".join(formatted_lines).strip()
99
+ return "\\n ".join(formatted_lines).strip()
100
100
 
101
101
 
102
102
  def create_ascii_safe_upload_copy(audio_file: str) -> tuple[str, str | None]:
@@ -156,7 +156,7 @@ class BeautifulTextConverter:
156
156
  finalize_nodes()
157
157
  return chapters
158
158
 
159
- def convert(self, raw_text: str, save_transcript_chunks: bool = False, active_chapters: bool = False) -> dict:
159
+ def convert(self, raw_text: str, save_transcript_chunks: bool = False, active_chapters: bool = True) -> dict:
160
160
  cleaned_input = (raw_text or "").strip()
161
161
  if not cleaned_input:
162
162
  result = {
@@ -204,6 +204,6 @@ class BeautifulTextConverter:
204
204
  "text_chunks": cleaned_chunks if save_transcript_chunks else "not provided",
205
205
  }
206
206
  if active_chapters:
207
- result["markdown_json"] = self._convert_markdown_to_json(final_text)
207
+ # result["markdown_json"] = self._convert_markdown_to_json(final_text)
208
208
  result["chapters"] = self._build_chapters(final_text)
209
209
  return result
@@ -13,6 +13,7 @@ from google.api_core import exceptions as google_exceptions
13
13
  from ..prompts.ocr import (
14
14
  OCR_TO_MARKDOWN_NON_LITERAL_FALLBACK_PROMPT,
15
15
  OCR_TO_MARKDOWN_PROMPT,
16
+ OCR_TO_PLAIN_TEXT_NON_LITERAL_FALLBACK_PROMPT,
16
17
  OCR_TO_PLAIN_TEXT_PROMPT,
17
18
  build_ocr_prompt,
18
19
  )
@@ -34,7 +35,7 @@ OCR_TAIL_REPETITION_THRESHOLD = float(os.getenv("OCR_TAIL_REPETITION_THRESHOLD",
34
35
  OCR_FALLBACK_SOURCE_PATTERN = os.getenv("OCR_FALLBACK_SOURCE_PATTERN", "flash-lite-preview")
35
36
  OCR_FALLBACK_MODEL = os.getenv("OCR_FALLBACK_MODEL", "gemini-3-flash-preview")
36
37
  OCR_FALLBACK_TEMPERATURE = float(os.getenv("OCR_FALLBACK_TEMPERATURE", "1.0"))
37
- OCR_FINAL_FALLBACK_MODEL = os.getenv("OCR_FINAL_FALLBACK_MODEL", "gemini-2.0-flash")
38
+ OCR_FINAL_FALLBACK_MODEL = os.getenv("OCR_FINAL_FALLBACK_MODEL", "gemini-3.5-flash")
38
39
  OCR_PROMPT_VARIANT_DEFAULT = "default"
39
40
  OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK = "non_literal_fallback"
40
41
  OCR_RETRIABLE_OUTPUT_ERROR_CODES = (996, 997, 999)
@@ -113,6 +114,7 @@ def get_document_ocr(
113
114
  ocr_model: str | None = None,
114
115
  max_output_tokens: int | None = None,
115
116
  include_image_descriptions: bool = False,
117
+ allow_partial_ocr_failures: bool = False,
116
118
  ):
117
119
  """
118
120
  Convenience function to extract text from an image file using OCR, optionally formatted as Markdown.
@@ -136,6 +138,9 @@ def get_document_ocr(
136
138
  include_image_descriptions (bool, optional): If True, OCR prompts include
137
139
  brief functional descriptions for meaningful non-text images.
138
140
  Defaults to False.
141
+ allow_partial_ocr_failures (bool, optional): If True, pages that still
142
+ fail OCR after all retries are recorded inline instead of aborting
143
+ the whole document extraction. Defaults to False.
139
144
 
140
145
  Returns:
141
146
  dict: Dictionary containing the OCR results and metadata.
@@ -149,6 +154,7 @@ def get_document_ocr(
149
154
  timeout_minutes=timeout_minutes,
150
155
  max_output_tokens=max_output_tokens,
151
156
  include_image_descriptions=include_image_descriptions,
157
+ allow_partial_ocr_failures=allow_partial_ocr_failures,
152
158
  )
153
159
  return converter.get_document_ocr(document_for_ocr)
154
160
 
@@ -157,7 +163,8 @@ class DocumentOCRToTextConverter:
157
163
  markdown_output=True, llm_api_key=None, target_size=1, temp_dir="temp",
158
164
  page_range=None, timeout_minutes: int = None, fallback_stage: int = 0,
159
165
  max_output_tokens: int | None = None, include_image_descriptions: bool = False,
160
- prompt_variant: str = OCR_PROMPT_VARIANT_DEFAULT):
166
+ prompt_variant: str = OCR_PROMPT_VARIANT_DEFAULT,
167
+ allow_partial_ocr_failures: bool = False):
161
168
  """
162
169
  Initialize the DocumentOCRToTextConverter class with specified OCR model and formatting options.
163
170
 
@@ -182,6 +189,9 @@ class DocumentOCRToTextConverter:
182
189
  Defaults to False.
183
190
  prompt_variant (str, optional): Prompt variant used by this attempt.
184
191
  Defaults to "default".
192
+ allow_partial_ocr_failures (bool, optional): If True, pages that still
193
+ fail OCR after all retries are recorded inline instead of aborting
194
+ the whole document extraction. Defaults to False.
185
195
 
186
196
  Raises:
187
197
  OSError: If temp directory creation fails
@@ -196,6 +206,7 @@ class DocumentOCRToTextConverter:
196
206
  self.timeout_minutes = timeout_minutes
197
207
  self.include_image_descriptions = include_image_descriptions
198
208
  self.prompt_variant = prompt_variant
209
+ self.allow_partial_ocr_failures = allow_partial_ocr_failures
199
210
  requested_output_tokens = OCR_MAX_OUTPUT_TOKENS if max_output_tokens is None else max_output_tokens
200
211
  self.max_output_tokens = max(requested_output_tokens, OCR_MIN_OUTPUT_TOKENS)
201
212
  self.fallback_stage = fallback_stage
@@ -212,6 +223,8 @@ class DocumentOCRToTextConverter:
212
223
  def _build_prompt_template(self) -> str:
213
224
  if self.markdown_output and self.prompt_variant == OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
214
225
  base_prompt = OCR_TO_MARKDOWN_NON_LITERAL_FALLBACK_PROMPT
226
+ elif not self.markdown_output and self.prompt_variant == OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
227
+ base_prompt = OCR_TO_PLAIN_TEXT_NON_LITERAL_FALLBACK_PROMPT
215
228
  elif self.markdown_output:
216
229
  base_prompt = OCR_TO_MARKDOWN_PROMPT
217
230
  else:
@@ -224,8 +237,6 @@ class DocumentOCRToTextConverter:
224
237
  def should_prompt_fallback_retry(self, error: EmptyDocument) -> bool:
225
238
  if self.fallback_stage != 0:
226
239
  return False
227
- if not self.markdown_output:
228
- return False
229
240
  if self.prompt_variant == OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
230
241
  return False
231
242
  return error.code in OCR_RETRIABLE_OUTPUT_ERROR_CODES
@@ -283,6 +294,7 @@ class DocumentOCRToTextConverter:
283
294
  max_output_tokens=self.max_output_tokens,
284
295
  include_image_descriptions=self.include_image_descriptions,
285
296
  prompt_variant=resolved_prompt_variant,
297
+ allow_partial_ocr_failures=self.allow_partial_ocr_failures,
286
298
  )
287
299
  result = fallback_converter.get_ocr(
288
300
  file_for_ocr=file_for_ocr,
@@ -541,7 +553,26 @@ class DocumentOCRToTextConverter:
541
553
  pix.save(temp_image_path)
542
554
 
543
555
  # Perform OCR on the page
544
- ocr_result = self.get_ocr(temp_image_path)
556
+ try:
557
+ ocr_result = self.get_ocr(temp_image_path)
558
+ except EmptyDocument as error:
559
+ if not self.allow_partial_ocr_failures:
560
+ raise
561
+ logger.warning(
562
+ "Document OCR failed on page %s after retries; keeping partial document because allow_partial_ocr_failures=True: %s",
563
+ page_num + 1,
564
+ error.message,
565
+ )
566
+ ocr_result = {
567
+ "text": "",
568
+ "completion_tokens": 0,
569
+ "prompt_tokens": 0,
570
+ "completion_model": self.ocr_model,
571
+ "completion_model_provider": self.ocr_model_provider,
572
+ "text_chunks": "not provided",
573
+ "page_error": True,
574
+ "page_error_reason": error.message,
575
+ }
545
576
  return page_num, ocr_result
546
577
 
547
578
  finally:
@@ -574,11 +605,17 @@ class DocumentOCRToTextConverter:
574
605
  all_text = []
575
606
  total_completion_tokens = 0
576
607
  total_prompt_tokens = 0
608
+ failed_pages = []
577
609
 
578
- for _, ocr_result in results:
610
+ for page_num, ocr_result in results:
579
611
  all_text.append(f"{ocr_result['text']}\n")
580
612
  total_completion_tokens += ocr_result['completion_tokens']
581
613
  total_prompt_tokens += ocr_result['prompt_tokens']
614
+ if ocr_result.get("page_error"):
615
+ failed_pages.append({
616
+ "page": page_num + 1,
617
+ "reason": ocr_result.get("page_error_reason", "unknown"),
618
+ })
582
619
 
583
620
  pdf.close()
584
621
 
@@ -589,6 +626,8 @@ class DocumentOCRToTextConverter:
589
626
  "completion_model": self.ocr_model,
590
627
  "completion_model_provider": self.ocr_model_provider,
591
628
  "text_chunks": "not provided",
629
+ "ocr_failed_pages": [item["page"] for item in failed_pages],
630
+ "ocr_failed_pages_detail": failed_pages,
592
631
  }
593
632
 
594
633
  return final_result_dict
@@ -13,6 +13,7 @@ from google.api_core import exceptions as google_exceptions
13
13
  from ..prompts.ocr import (
14
14
  OCR_TO_MARKDOWN_NON_LITERAL_FALLBACK_PROMPT,
15
15
  OCR_TO_MARKDOWN_PROMPT,
16
+ OCR_TO_PLAIN_TEXT_NON_LITERAL_FALLBACK_PROMPT,
16
17
  OCR_TO_PLAIN_TEXT_PROMPT,
17
18
  build_ocr_prompt,
18
19
  )
@@ -207,6 +208,8 @@ class OCRToTextConverter:
207
208
  def _build_prompt_template(self) -> str:
208
209
  if self.markdown_output and self.prompt_variant == OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
209
210
  base_prompt = OCR_TO_MARKDOWN_NON_LITERAL_FALLBACK_PROMPT
211
+ elif not self.markdown_output and self.prompt_variant == OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
212
+ base_prompt = OCR_TO_PLAIN_TEXT_NON_LITERAL_FALLBACK_PROMPT
210
213
  elif self.markdown_output:
211
214
  base_prompt = OCR_TO_MARKDOWN_PROMPT
212
215
  else:
@@ -219,8 +222,6 @@ class OCRToTextConverter:
219
222
  def should_prompt_fallback_retry(self, error: EmptyDocument) -> bool:
220
223
  if self.fallback_stage != 0:
221
224
  return False
222
- if not self.markdown_output:
223
- return False
224
225
  if self.prompt_variant == OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
225
226
  return False
226
227
  return error.code in OCR_RETRIABLE_OUTPUT_ERROR_CODES
@@ -90,7 +90,9 @@ def _raise_empty_document_loader_error(error: EmptyDocument) -> None:
90
90
  class BaseLoader:
91
91
  def __init__(self, markdown_output=True, llm_api_key=None, provider: str = "google", temp_dir: str = "temp",
92
92
  ocr_model: str = "gpt-5-mini", timeout_minutes: int | None = None,
93
- include_image_descriptions: bool | None = None, **kwargs):
93
+ include_image_descriptions: bool | None = None,
94
+ force_ocr: bool = False,
95
+ **kwargs):
94
96
  """
95
97
  Initialize the BaseLoader with cloud storage and LLM configurations.
96
98
 
@@ -108,6 +110,8 @@ class BaseLoader:
108
110
  include_image_descriptions (bool | None, optional): If True, OCR prompts
109
111
  include brief functional descriptions for meaningful non-text images.
110
112
  If None, defaults from OCR_INCLUDE_IMAGE_DESCRIPTIONS. Defaults to None.
113
+ force_ocr (bool, optional): If True, supported document files are routed to
114
+ OCRLoader instead of the standard DocumentLoader. Defaults to False.
111
115
  **kwargs: Additional keyword arguments to pass to the underlying loader or extraction logic.
112
116
  - target_size (int, optional): Target file size in bytes. Defaults to 1MB
113
117
  - source (str): Source of the document. Must be either "cloud" or "local"
@@ -131,6 +135,7 @@ class BaseLoader:
131
135
  if include_image_descriptions is None
132
136
  else include_image_descriptions
133
137
  )
138
+ self.force_ocr = force_ocr
134
139
  self.kwargs = kwargs
135
140
  self.target_size = kwargs.get("target_size", 1)
136
141
  self.source = kwargs.get("source", "cloud")
@@ -259,8 +264,14 @@ class BaseLoader:
259
264
  cleanup_result = converter.convert(
260
265
  raw_text=raw_result["text"],
261
266
  save_transcript_chunks=kwargs.get("save_transcript_chunks", self.save_transcript_chunks),
262
- active_chapters=kwargs.get("active_chapters", False),
267
+ active_chapters=kwargs.get("active_chapters", True),
263
268
  )
269
+ if not cleanup_result.get("chapters"):
270
+ raise LoaderError(
271
+ message="No chapters detected",
272
+ status=422,
273
+ code="NO_CHAPTERS_DETECTED",
274
+ )
264
275
 
265
276
  total_completion_tokens = raw_result.get("completion_tokens", 0) + cleanup_result.get("completion_tokens", 0)
266
277
  total_prompt_tokens = raw_result.get("prompt_tokens", 0) + cleanup_result.get("prompt_tokens", 0)
@@ -405,8 +416,7 @@ class BaseLoader:
405
416
  if path_without_query:
406
417
  _, file_extension = os.path.splitext(path_without_query)
407
418
  else: # If is local file path (without schema)
408
- if os.path.exists(input):
409
- _, file_extension = os.path.splitext(input)
419
+ _, file_extension = os.path.splitext(input)
410
420
 
411
421
  if file_extension:
412
422
  file_extension = file_extension.lower()
@@ -436,7 +446,29 @@ class BaseLoader:
436
446
  )
437
447
  elif mime_type:
438
448
  if file_extension in [".pdf", ".xlsx", ".docx", ".txt", ".csv", ".odt", ".pptx", ".xls", ".doc", ".ppt", ".rtf"]:
439
- return DocumentLoader(markdown_output=self.markdown_output, temp_dir=self.temp_dir, timeout_minutes=self.timeout_minutes, **kwargs)
449
+ document_kwargs = {k: v for k, v in kwargs.items() if k != "source"}
450
+ if self.force_ocr:
451
+
452
+ return DocumentOCRLoader(
453
+ source=self.source,
454
+ llm_api_key=llm_api_key,
455
+ markdown_output=self.markdown_output,
456
+ temp_dir=self.temp_dir,
457
+ timeout_minutes=self.timeout_minutes,
458
+ ocr_provider=self.provider,
459
+ ocr_model=self.ocr_model,
460
+ include_image_descriptions=self.include_image_descriptions,
461
+ allow_partial_ocr_failures=True,
462
+ **document_kwargs,
463
+ )
464
+
465
+ return DocumentLoader(
466
+ source=self.source,
467
+ markdown_output=self.markdown_output,
468
+ temp_dir=self.temp_dir,
469
+ timeout_minutes=self.timeout_minutes,
470
+ **document_kwargs,
471
+ )
440
472
  elif mime_type.startswith("audio/"):
441
473
  audio_kwargs = {**kwargs, "is_output_audio_raw": self.is_output_audio_raw}
442
474
  return AudioLoader(llm_api_key=llm_api_key, markdown_output=self.markdown_output, temp_dir=self.temp_dir, timeout_minutes=self.timeout_minutes, **audio_kwargs)
@@ -565,14 +597,18 @@ class BaseLoader:
565
597
 
566
598
  result_dict["text"] = clean_extracted_text_whitespace(remove_markdown_strip(result_dict["text"]))
567
599
 
568
- result_dict = {
600
+ final_result = {
569
601
  "text": result_dict["text"],
570
602
  "completion_tokens": result_dict["completion_tokens"],
571
603
  "prompt_tokens": result_dict["prompt_tokens"],
572
604
  "output_list": [result_dict],
573
605
  }
574
606
 
575
- return result_dict
607
+ for metadata_key in ("ocr_failed_pages", "ocr_failed_pages_detail"):
608
+ if metadata_key in result_dict:
609
+ final_result[metadata_key] = result_dict[metadata_key]
610
+
611
+ return final_result
576
612
 
577
613
  # Helper methods
578
614
  @staticmethod
@@ -46,6 +46,7 @@ class DocumentOCRLoader:
46
46
  ocr_provider: str = "google",
47
47
  ocr_model: str | None = None,
48
48
  include_image_descriptions: bool = False,
49
+ allow_partial_ocr_failures: bool = False,
49
50
  **kwargs
50
51
  ):
51
52
  """
@@ -82,6 +83,9 @@ class DocumentOCRLoader:
82
83
  include_image_descriptions (bool, optional): If True, OCR prompts include
83
84
  brief functional descriptions for meaningful non-text images.
84
85
  Defaults to False.
86
+ allow_partial_ocr_failures (bool, optional): If True, pages that still
87
+ fail OCR after all retries are recorded inline instead of aborting
88
+ the whole document extraction. Defaults to False.
85
89
  **kwargs:
86
90
  max_output_tokens (int, optional): Maximum Gemini output tokens for
87
91
  Google document OCR generation.
@@ -105,6 +109,7 @@ class DocumentOCRLoader:
105
109
  self.ocr_provider = (ocr_provider or "google").lower()
106
110
  self.ocr_model = ocr_model
107
111
  self.include_image_descriptions = include_image_descriptions
112
+ self.allow_partial_ocr_failures = allow_partial_ocr_failures
108
113
  self.max_output_tokens = kwargs.get("max_output_tokens")
109
114
 
110
115
  # Set up custom temp directory
@@ -254,6 +259,7 @@ class DocumentOCRLoader:
254
259
  timeout_minutes=self.timeout_minutes,
255
260
  ocr_model=self.ocr_model or None,
256
261
  include_image_descriptions=self.include_image_descriptions,
262
+ allow_partial_ocr_failures=self.allow_partial_ocr_failures,
257
263
  )
258
264
  else:
259
265
  result_dict = ocr_fn(
@@ -270,6 +276,7 @@ class DocumentOCRLoader:
270
276
  ),
271
277
  max_output_tokens=self.max_output_tokens,
272
278
  include_image_descriptions=self.include_image_descriptions,
279
+ allow_partial_ocr_failures=self.allow_partial_ocr_failures,
273
280
  )
274
281
 
275
282
  result_dict["type"] = self.type
@@ -0,0 +1,61 @@
1
+ BEAUTIFUL_TEXT_PROMPT = """
2
+ You are cleaning a spoken transcript into faithful Markdown.
3
+
4
+ This is NOT summarization.
5
+ This is NOT rewriting.
6
+ This is NOT editorial adaptation.
7
+ This is a cleaned transcript.
8
+
9
+ Your task is to remove only clearly meaningless noise while preserving the original spoken wording,
10
+ sequence, tone, rhythm, and reasoning as closely as possible.
11
+
12
+ REMOVE ONLY:
13
+ - obvious filler sounds such as "eh", "uhm", "mmh" when they clearly add no meaning
14
+ - accidental duplicated words such as "di di", "che che", "da da"
15
+ - clearly aborted false starts that add no meaning
16
+ - irrelevant overlap fragments between speakers
17
+
18
+ IF THERE IS ANY DOUBT, KEEP THE ORIGINAL WORDING.
19
+
20
+ PRESERVE STRICTLY:
21
+ - the original wording
22
+ - the original order of ideas
23
+ - the original paragraph flow
24
+ - the original tone and conversational style
25
+ - repetitions that still carry emphasis, rhythm, hesitation, or meaning
26
+ - colloquial phrasing and spoken transitions when meaningful
27
+
28
+ DO NOT:
29
+ - summarize
30
+ - compress
31
+ - simplify
32
+ - polish into formal written prose
33
+ - merge multiple spoken sentences into a shorter reformulation
34
+ - turn the transcript into an article, essay, report, or explanatory text
35
+ - replace words with better synonyms
36
+ - add explanations, transitions, or inferred content
37
+ - add interpretive conclusions
38
+
39
+ FORMATTING:
40
+ - output Markdown only
41
+ - preserve the transcript as a cleaned spoken transcript, not as a rewritten article
42
+ - use paragraphs, but do not heavily reorganize the flow
43
+ - headings may be generated editorially, but only to label the topic of the following block
44
+ - headings must be short, neutral, and strictly supported by the text below
45
+ - the final Markdown must contain at least one heading
46
+ - if the text is short or has weak structure, add one minimal heading only
47
+ - do not convert prose into bullet lists unless the speaker is explicitly enumerating points
48
+ - use emphasis sparingly
49
+ - you may add light Markdown emphasis to improve scanability, but only locally and without rewriting the sentence
50
+ - use **bold** for clearly salient entities already present in the source, such as names, products, platforms, institutions, laws, or central concepts
51
+ - use *italics* sparingly for contextual labels, foreign expressions, or technical terms when this remains clearly faithful to the source
52
+ - do not apply emphasis to large portions of text
53
+ - do not use emphasis as a substitute for rewriting, summarizing, or restructuring
54
+ - do not add emphasis just to make the text nicer
55
+ - do not add code fences or commentary
56
+
57
+ FINAL CHECK:
58
+ - every output sentence must remain closely traceable to the input
59
+ - prefer awkward fidelity over elegant rewriting
60
+ - headings may be editorially generated, but body text must remain maximally faithful
61
+ """
@@ -28,6 +28,15 @@ Maintain paragraph breaks and formatting.
28
28
  Your output must be a plain text.
29
29
  """
30
30
 
31
+ OCR_TO_PLAIN_TEXT_NON_LITERAL_FALLBACK_PROMPT = """
32
+ Rephrase and reorganize the text content in a coherent plain text structure.
33
+ Do not transcribe the text verbatim, but preserve the meaning of the original content without omitting anything, even apparently minor details.
34
+ Pay special attention to tables, columns, headers, and any structured content.
35
+ Maintain paragraph breaks and formatting.
36
+ Your output must be a plain text.
37
+ In case no readable text is present, write exactly "no readable text present".
38
+ """
39
+
31
40
  OCR_IMAGE_DESCRIPTION_INSTRUCTIONS = """
32
41
  Image description instructions:
33
42
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: polytext
3
- Version: 0.2.8
3
+ Version: 0.2.8b2
4
4
  Summary: Python utilities to simplify document files management
5
5
  Home-page: https://github.com/docsity/polytext
6
6
  Author: Matteo Senardi
@@ -51,7 +51,7 @@ def get_requirements(*requirements_file):
51
51
 
52
52
  setup(
53
53
  name='polytext',
54
- version='0.2.8',
54
+ version='0.2.8b2',
55
55
  url='https://github.com/docsity/polytext',
56
56
  # download_url='https://github.com/pualien/py-polytext/archive/0.1.23.tar.gz',
57
57
  license='MIT',
@@ -137,7 +137,7 @@ class TestAudioTranscriptionModelMigration(unittest.TestCase):
137
137
 
138
138
  self.assertEqual(
139
139
  formatted,
140
- "Prima frase.\nSeconda frase?\nTerza frase!\n## Titolo\nQuarta frase.\nQuinta frase.",
140
+ "Prima frase.\\n Seconda frase?\\n Terza frase!\\n ## Titolo\\n Quarta frase.\\n Quinta frase.",
141
141
  )
142
142
 
143
143
  def test_normalize_no_human_speech_marker_returns_empty_for_marker_only(self):
@@ -1,4 +1,5 @@
1
1
  import unittest
2
+ from types import SimpleNamespace
2
3
  from unittest.mock import Mock, patch
3
4
 
4
5
  from polytext.exceptions import ConversionError, EmptyDocument, LoaderError
@@ -40,6 +41,24 @@ class _FallbackFailingBaseLoader(BaseLoader):
40
41
  return _FailingLoader(self.initial_error)
41
42
 
42
43
 
44
+ class _BeautifulTextLoader(BaseLoader):
45
+ def __init__(self, raw_result=None, **kwargs):
46
+ super().__init__(**kwargs)
47
+ self.raw_result = raw_result or {
48
+ "text": "raw text",
49
+ "completion_tokens": 0,
50
+ "prompt_tokens": 0,
51
+ "completion_model": "not provided",
52
+ "completion_model_provider": "not provided",
53
+ "text_chunks": "not provided",
54
+ "type": "text",
55
+ "input": "dummy input",
56
+ }
57
+
58
+ def extract_raw_text_for_beautiful_text(self, input_value: str, **kwargs) -> dict:
59
+ return self.raw_result
60
+
61
+
43
62
  class TestBaseLoaderErrorMapping(unittest.TestCase):
44
63
  def test_llm_output_empty_document_codes_are_raised_as_loader_errors(self):
45
64
  cases = [
@@ -143,6 +162,34 @@ class TestBaseLoaderErrorMapping(unittest.TestCase):
143
162
  mock_exception.assert_not_called()
144
163
  sentry_sdk.capture_exception.assert_called_once_with(conversion_error)
145
164
 
165
+ def test_beautiful_text_without_detected_chapters_is_raised_as_loader_error(self):
166
+ loader = _BeautifulTextLoader()
167
+ cleanup_result = {
168
+ "text": "Short text without headings.",
169
+ "completion_tokens": 1,
170
+ "prompt_tokens": 1,
171
+ "completion_model": "gemini-3.1-flash-lite",
172
+ "completion_model_provider": "google",
173
+ "text_chunks": "not provided",
174
+ "markdown_json": {"root": ["Short text without headings."]},
175
+ "chapters": [],
176
+ }
177
+
178
+ fake_converter_class = Mock()
179
+ fake_converter_class.return_value.convert.return_value = cleanup_result
180
+
181
+ with patch.dict(
182
+ "sys.modules",
183
+ {"polytext.converter.beautiful_text": SimpleNamespace(BeautifulTextConverter=fake_converter_class)},
184
+ ):
185
+ with self.assertRaises(LoaderError) as error_context:
186
+ loader.get_beautiful_text(["dummy input"], active_chapters=True)
187
+
188
+ error = error_context.exception
189
+ self.assertEqual(error.status, 422)
190
+ self.assertEqual(error.code, "NO_CHAPTERS_DETECTED")
191
+ self.assertEqual(error.message, "No chapters detected")
192
+
146
193
 
147
194
  if __name__ == "__main__":
148
195
  unittest.main()
@@ -1,43 +0,0 @@
1
- BEAUTIFUL_TEXT_PROMPT = """
2
- You are an editor specialized in cleaning spoken transcripts and raw text into faithful Markdown.
3
- This is not summarization. This is not rewriting. This is a cleaned transcript or cleaned source text.
4
-
5
- Your task is to remove only accidental noise while preserving the speaker's or author's original words,
6
- phrasing, reasoning, tone, and sequence of ideas as faithfully as possible.
7
-
8
- REMOVE ONLY:
9
- - non-meaningful fillers such as "eh", "uhm", "diciamo", "eccetera eccetera", "no?" when used only as filler
10
- - redundant "quindi", "appunto", "comunque" when they are only conversational padding
11
- - accidental repeated words such as "di di", "da da", "che che"
12
- - false starts and self-corrections only when they do not carry meaning
13
- - irrelevant overlap fragments between speakers
14
-
15
- PRESERVE COMPLETELY:
16
- - the original wording and sentence structure, even if colloquial
17
- - technical terms and proper nouns exactly
18
- - the original tone and register
19
- - reasoning, opinions, nuances, and meaningful uncertainty
20
- - the logical order of the discussion
21
-
22
- DO NOT:
23
- - rewrite sentences in a more elegant style
24
- - replace words with synonyms
25
- - summarize, compress, or simplify concepts
26
- - add explanations, transitions, or missing content
27
- - correct the speaker's opinions or inaccuracies
28
- - make the language more formal than the original
29
-
30
- FORMATTING:
31
- - output Markdown only
32
- - use paragraphs to separate thematic blocks
33
- - add headings only when the speaker explicitly introduces a new topic
34
- - use bullet lists or numbered lists only when the source explicitly enumerates items or when the sequence is clearly list-shaped
35
- - use emphasis sparingly and only when grounded in the original text
36
- - use **bold** for key information and important concepts, and *italics* for subtle emphasis or contextual terms in every chapter and paragraph whenever they improve readability and understanding
37
- - do not add code fences
38
- - do not add introductions or commentary
39
-
40
- FINAL CHECK:
41
- - every sentence in the output must be traceable to an equivalent sentence in the input
42
- - if a sentence cannot be grounded in the input, remove it
43
- """
File without changes
File without changes
File without changes
File without changes