polytext 0.2.8b3__tar.gz → 0.2.8b4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. {polytext-0.2.8b3 → polytext-0.2.8b4}/PKG-INFO +2 -2
  2. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/audio_to_text.py +213 -19
  3. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/beautiful_text.py +63 -14
  4. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/document_ocr_to_text.py +5 -1
  5. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/gemini_quality_guards.py +8 -0
  6. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/ocr_to_text.py +2 -0
  7. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/text_to_md.py +3 -1
  8. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/video_to_audio.py +4 -6
  9. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/base.py +0 -1
  10. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/youtube_llm.py +2 -0
  11. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/processor/audio_chunker.py +4 -5
  12. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/processor/text_merger.py +56 -30
  13. polytext-0.2.8b4/polytext/prompts/beautiful_text.py +97 -0
  14. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/prompts/transcription.py +32 -12
  15. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext.egg-info/PKG-INFO +2 -2
  16. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext.egg-info/requires.txt +1 -1
  17. {polytext-0.2.8b3 → polytext-0.2.8b4}/setup.py +1 -1
  18. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_audio_chunker.py +8 -2
  19. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_audio_transcription_model_migration.py +333 -22
  20. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_ocr_fallbacks.py +32 -0
  21. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_ocr_image_descriptions.py +13 -0
  22. polytext-0.2.8b3/polytext/prompts/beautiful_text.py +0 -61
  23. {polytext-0.2.8b3 → polytext-0.2.8b4}/LICENSE +0 -0
  24. {polytext-0.2.8b3 → polytext-0.2.8b4}/README.md +0 -0
  25. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/__init__.py +0 -0
  26. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/__init__.py +0 -0
  27. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/base.py +0 -0
  28. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/document_ocr_to_text_azure_oai.py +0 -0
  29. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/html_to_md.py +0 -0
  30. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/md_to_text.py +0 -0
  31. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/ocr_to_text_azure_oai.py +0 -0
  32. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/pdf.py +0 -0
  33. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/exceptions/__init__.py +0 -0
  34. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/exceptions/base.py +0 -0
  35. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/generator/__init__.py +0 -0
  36. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/generator/pdf.py +0 -0
  37. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/__init__.py +0 -0
  38. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/audio.py +0 -0
  39. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/aws_auth.py +0 -0
  40. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/document.py +0 -0
  41. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/document_ocr.py +0 -0
  42. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/downloader/__init__.py +0 -0
  43. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/downloader/downloader.py +0 -0
  44. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/html.py +0 -0
  45. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/markdown.py +0 -0
  46. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/notebook.py +0 -0
  47. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/ocr.py +0 -0
  48. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/plain_text.py +0 -0
  49. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/video.py +0 -0
  50. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/xml_xbrl.py +0 -0
  51. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/youtube.py +0 -0
  52. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/processor/__init__.py +0 -0
  53. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/processor/transcript_chunker.py +0 -0
  54. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/prompts/__init__.py +0 -0
  55. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/prompts/ocr.py +0 -0
  56. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/prompts/text_merging.py +0 -0
  57. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/prompts/text_to_md.py +0 -0
  58. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/utils/__init__.py +0 -0
  59. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/utils/utils.py +0 -0
  60. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext.egg-info/SOURCES.txt +0 -0
  61. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext.egg-info/dependency_links.txt +0 -0
  62. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext.egg-info/not-zip-safe +0 -0
  63. {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext.egg-info/top_level.txt +0 -0
  64. {polytext-0.2.8b3 → polytext-0.2.8b4}/pyproject.toml +0 -0
  65. {polytext-0.2.8b3 → polytext-0.2.8b4}/setup.cfg +0 -0
  66. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_audio_comparison_helpers.py +0 -0
  67. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_aws_auth.py +0 -0
  68. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_base_loader_error_mapping.py +0 -0
  69. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_beautiful_text_manual.py +0 -0
  70. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_compare_audio_models.py +0 -0
  71. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_compare_document_ocr_to_text_models.py +0 -0
  72. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_compare_ocr_to_text_models.py +0 -0
  73. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_compare_youtube_models.py +0 -0
  74. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_dowload_audio_from_youtube.py +0 -0
  75. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_dowload_audio_from_youtube_helpers.py +0 -0
  76. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_extracted_text_whitespace.py +0 -0
  77. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_gemini_quality_guards.py +0 -0
  78. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_audio_transcript_from_gcs.py +0 -0
  79. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_customized_pdf_from_markdown.py +0 -0
  80. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_document_ocr.py +0 -0
  81. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_document_ocr_azure_oai.py +0 -0
  82. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_document_text.py +0 -0
  83. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_document_text_from_gcs.py +0 -0
  84. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_ocr_from_image.py +0 -0
  85. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_text_from_markdown.py +0 -0
  86. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_video_transcript_from_gcs.py +0 -0
  87. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_library.py +0 -0
  88. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_markdown_loader_gzip.py +0 -0
  89. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_markitdown_html.py +0 -0
  90. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_notebook_loader.py +0 -0
  91. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_pain_text.py +0 -0
  92. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_pdf_conversion_error.py +0 -0
  93. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_python_version_metadata.py +0 -0
  94. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_split_audio_with_llm.py +0 -0
  95. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_transcribe_s3_images_from_csv.py +0 -0
  96. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_transcribe_s3_images_from_csv_script.py +0 -0
  97. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_xml_xbrl_loader.py +0 -0
  98. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_youtube_gemini_minimal_check.py +0 -0
  99. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_youtube_llm_fallbacks.py +0 -0
  100. {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_youtube_transcript.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: polytext
3
- Version: 0.2.8b3
3
+ Version: 0.2.8b4
4
4
  Summary: Python utilities to simplify document files management
5
5
  Home-page: https://github.com/docsity/polytext
6
6
  Author: Matteo Senardi
@@ -25,7 +25,7 @@ Requires-Dist: markdown-to-json==2.1.2
25
25
  Requires-Dist: python-docx==1.1.2
26
26
  Requires-Dist: google-api-core>=2.24.2
27
27
  Requires-Dist: google-cloud-storage<3.0.0,>=2.17
28
- Requires-Dist: google-genai>=1.16.1
28
+ Requires-Dist: google-genai==2.22.0
29
29
  Requires-Dist: openai==2.26.0
30
30
  Requires-Dist: boto3>=1.42.64
31
31
  Requires-Dist: botocore>=1.42.64
@@ -14,6 +14,7 @@ from google.genai import types
14
14
  from google.genai import errors as genai_errors
15
15
  from concurrent.futures import ThreadPoolExecutor, as_completed
16
16
  from google.api_core import exceptions as google_exceptions
17
+ from pydub import AudioSegment
17
18
 
18
19
  from ..exceptions import EmptyDocument
19
20
  from ..prompts.transcription import (
@@ -25,13 +26,22 @@ from ..prompts.transcription import (
25
26
  )
26
27
  from ..processor.audio_chunker import AudioChunker
27
28
  from ..processor.text_merger import TextMerger
28
- from .gemini_quality_guards import extract_finish_reason, tail_has_excessive_repetition
29
+ from .gemini_quality_guards import (
30
+ extract_finish_reason,
31
+ has_excessive_consecutive_word_repetition,
32
+ tail_has_excessive_repetition,
33
+ )
29
34
 
30
35
  logger = logging.getLogger(__name__)
31
36
 
32
37
  SUPPORTED_MIME_TYPES = {
33
38
  'audio/x-aac', 'audio/flac', 'audio/mp3', 'audio/m4a', 'audio/mpeg',
34
- 'audio/mpga', 'audio/mp4', 'audio/opus', 'audio/pcm', 'audio/wav', 'audio/webm'
39
+ 'audio/mpga', 'audio/mp4', 'audio/opus', 'audio/pcm', 'audio/wav', 'audio/x-wav', 'audio/webm'
40
+ }
41
+
42
+ GEMINI_AUDIO_MIME_ALIASES = {
43
+ 'audio/x-aac': 'audio/aac',
44
+ 'audio/x-wav': 'audio/wav',
35
45
  }
36
46
 
37
47
  INJECTION_GUARD_SYSTEM_INSTRUCTION = (
@@ -48,16 +58,20 @@ INJECTION_GUARD_SYSTEM_INSTRUCTION = (
48
58
  )
49
59
 
50
60
  AUDIO_MIN_OUTPUT_TOKENS = 500
61
+ AUDIO_DEFAULT_MAX_OUTPUT_TOKENS = 4096
62
+ AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH = 1
63
+ AUDIO_ADAPTIVE_SPLIT_OVERLAP_MS = 2000
64
+ AUDIO_LONG_DURATION_THRESHOLD_MS = 80 * 60 * 1000
51
65
  AUDIO_TAIL_REPETITION_LINES = int(os.getenv("AUDIO_TAIL_REPETITION_LINES", "200"))
52
66
  AUDIO_TAIL_REPETITION_THRESHOLD = float(os.getenv("AUDIO_TAIL_REPETITION_THRESHOLD", "0.35"))
53
67
  AUDIO_FALLBACK_SOURCE_PATTERN = os.getenv("AUDIO_FALLBACK_SOURCE_PATTERN", "flash-lite")
54
- AUDIO_FALLBACK_MODEL = os.getenv("AUDIO_FALLBACK_MODEL", "gemini-3-flash-preview")
68
+ AUDIO_FALLBACK_MODEL = os.getenv("AUDIO_FALLBACK_MODEL", "gemini-3.5-flash-lite")
55
69
  AUDIO_FALLBACK_TEMPERATURE = float(os.getenv("AUDIO_FALLBACK_TEMPERATURE", "1.0"))
56
70
  AUDIO_FINAL_FALLBACK_MODEL = os.getenv("AUDIO_FINAL_FALLBACK_MODEL", "gemini-3.5-flash")
57
71
  AUDIO_FILE_UPLOAD_THRESHOLD_BYTES = 20 * 1024 * 1024
58
72
  AUDIO_PROMPT_VARIANT_DEFAULT = "default"
59
73
  AUDIO_PROMPT_VARIANT_NON_LITERAL_FALLBACK = "non_literal_fallback"
60
- AUDIO_RETRIABLE_OUTPUT_ERROR_CODES = (996, 997, 999)
74
+ AUDIO_RETRIABLE_OUTPUT_ERROR_CODES = (996, 997, 998, 999)
61
75
  NO_HUMAN_SPEECH_MARKER = "no human speech detected"
62
76
 
63
77
 
@@ -123,33 +137,32 @@ def create_ascii_safe_upload_copy(audio_file: str) -> tuple[str, str | None]:
123
137
 
124
138
  def compress_and_convert_audio(input_path: str, bitrate_quality: int = 9) -> str:
125
139
  """
126
- Compress and convert an audio file to MP3 using ffmpeg.
140
+ Normalize an audio file to lossless 16 kHz mono WAV using ffmpeg.
127
141
 
128
142
  Args:
129
143
  input_path (str): Path to the original audio file
130
- bitrate_quality (int, optional): Variable bitrate quality from 0-9 (9 being lowest). Defaults to 9
144
+ bitrate_quality (int, optional): Retained for backward compatibility; WAV output is lossless.
131
145
 
132
146
  Returns:
133
- str: Path to the temporary compressed/converted MP3 file
147
+ str: Path to the temporary normalized WAV file
134
148
 
135
149
  Raises:
136
150
  RuntimeError: If FFmpeg compression/conversion fails
137
151
 
138
152
  Notes:
139
- - Creates a temporary MP3 file that should be deleted after use
140
- - Converts audio to mono and 16kHz sample rate for smaller file size
153
+ - Creates a temporary WAV file that should be deleted after use
154
+ - Converts audio to 16-bit PCM mono at 16kHz
141
155
  - Uses maximum available CPU threads for faster processing
142
156
  """
143
157
  # Create temporary file for audio output
144
- fd, temp_audio_path = tempfile.mkstemp(suffix='.mp3')
158
+ fd, temp_audio_path = tempfile.mkstemp(suffix='.wav')
145
159
  os.close(fd)
146
160
 
147
- logger.info(f"Compressing audio to bitrate quality: {bitrate_quality}")
161
+ logger.info("Normalizing audio to lossless 16 kHz mono WAV")
148
162
  try:
149
163
  ffmpeg.input(input_path).output(
150
164
  temp_audio_path,
151
- q=bitrate_quality, # Variable bitrate quality (0-9, 9 being lowest)
152
- acodec='libmp3lame',
165
+ acodec='pcm_s16le',
153
166
  ac=1, # Convert to mono
154
167
  ar=16000, # Lower sample rate
155
168
  vn=None,
@@ -162,7 +175,7 @@ def compress_and_convert_audio(input_path: str, bitrate_quality: int = 9) -> str
162
175
  os.unlink(temp_audio_path)
163
176
  raise
164
177
 
165
- logger.info(f"Successfully converted and compressed audio: {temp_audio_path}")
178
+ logger.info(f"Successfully normalized audio: {temp_audio_path}")
166
179
  return temp_audio_path
167
180
 
168
181
 
@@ -189,7 +202,7 @@ def transcribe_full_audio(audio_file, markdown_output: bool = False,
189
202
  bitrate_quality (int, optional): Variable bitrate quality from 0-9 (9 being lowest). Defaults to 9
190
203
  timeout_minutes (int, optional): Number of minutes to wait for a response. Defaults to None.
191
204
  max_llm_tokens (int, optional): Token budget used for audio chunk sizing. Defaults to 4250.
192
- max_output_tokens (int | None, optional): Maximum Gemini output tokens. Defaults to `max_llm_tokens`.
205
+ max_output_tokens (int | None, optional): Maximum Gemini output tokens. Defaults to 4096.
193
206
  is_output_audio_raw (bool, optional): If True, use the raw Markdown audio prompt.
194
207
  If False, use the formatted Markdown audio prompt. Defaults to True.
195
208
 
@@ -211,6 +224,8 @@ class AudioToTextConverter:
211
224
  max_output_tokens: int | None = None, temp_dir: str = "temp",
212
225
  bitrate_quality: int = 9, timeout_minutes: int = None,
213
226
  fallback_stage: int = 0,
227
+ adaptive_split_depth: int = 0,
228
+ long_audio_protections_enabled: bool = False,
214
229
  prompt_variant: str = AUDIO_PROMPT_VARIANT_DEFAULT,
215
230
  is_output_audio_raw: bool = True):
216
231
  """
@@ -225,12 +240,14 @@ class AudioToTextConverter:
225
240
  llm_api_key (str, optional): Override API key for language model. Defaults to None.
226
241
  max_llm_tokens (int): Token budget used to size audio chunks. Defaults to 4250.
227
242
  max_output_tokens (int | None): Maximum number of output tokens for Gemini generation.
228
- Defaults to `max_llm_tokens`.
243
+ Defaults to 4096.
229
244
  temp_dir (str): Directory for temporary files. Defaults to "temp".
230
245
  bitrate_quality (int, optional): Variable bitrate quality from 0-9 (9 being lowest). Defaults to 9
231
246
  timeout_minutes (int): Number of minutes to wait for a response.
232
247
  fallback_stage (int, optional): Internal retry stage used by fallback attempts.
233
248
  Defaults to 0.
249
+ long_audio_protections_enabled (bool, optional): Apply stricter recovery rules inherited
250
+ from an original audio longer than 80 minutes. Defaults to False.
234
251
  prompt_variant (str, optional): Prompt variant used by this attempt.
235
252
  Defaults to "default".
236
253
  is_output_audio_raw (bool, optional): If True, use the raw Markdown audio prompt.
@@ -247,12 +264,18 @@ class AudioToTextConverter:
247
264
  self.markdown_output = markdown_output
248
265
  self.llm_api_key = llm_api_key
249
266
  self.max_llm_tokens = max(max_llm_tokens, AUDIO_MIN_OUTPUT_TOKENS)
250
- requested_output_tokens = self.max_llm_tokens if max_output_tokens is None else max_output_tokens
267
+ requested_output_tokens = (
268
+ AUDIO_DEFAULT_MAX_OUTPUT_TOKENS
269
+ if max_output_tokens is None
270
+ else max_output_tokens
271
+ )
251
272
  self.max_output_tokens = max(requested_output_tokens, AUDIO_MIN_OUTPUT_TOKENS)
252
273
  self.chunked_audio = False
253
274
  self.bitrate_quality = bitrate_quality
254
275
  self.timeout_minutes = timeout_minutes
255
276
  self.fallback_stage = fallback_stage
277
+ self.adaptive_split_depth = adaptive_split_depth
278
+ self.long_audio_protections_enabled = long_audio_protections_enabled
256
279
  self.prompt_variant = prompt_variant
257
280
  self.fallback_source_pattern = AUDIO_FALLBACK_SOURCE_PATTERN
258
281
  self.fallback_model = AUDIO_FALLBACK_MODEL
@@ -276,6 +299,9 @@ class AudioToTextConverter:
276
299
  return AUDIO_TO_MARKDOWN_PROMPT
277
300
  return AUDIO_TO_PLAIN_TEXT_PROMPT
278
301
 
302
+ def set_long_audio_protections(self, duration_ms: int) -> None:
303
+ self.long_audio_protections_enabled = duration_ms > AUDIO_LONG_DURATION_THRESHOLD_MS
304
+
279
305
  def should_prompt_fallback_retry(self, error: EmptyDocument) -> bool:
280
306
  if self.fallback_stage != 0:
281
307
  return False
@@ -338,6 +364,8 @@ class AudioToTextConverter:
338
364
  bitrate_quality=self.bitrate_quality,
339
365
  timeout_minutes=self.timeout_minutes,
340
366
  fallback_stage=fallback_stage,
367
+ adaptive_split_depth=self.adaptive_split_depth,
368
+ long_audio_protections_enabled=self.long_audio_protections_enabled,
341
369
  prompt_variant=resolved_prompt_variant,
342
370
  is_output_audio_raw=self.is_output_audio_raw,
343
371
  )
@@ -353,10 +381,83 @@ class AudioToTextConverter:
353
381
  result.setdefault("fallback_to_prompt_variant", resolved_prompt_variant)
354
382
  return result
355
383
 
384
+ def transcribe_audio_halves(self, audio_file: str, temperature: float = 0.0) -> dict:
385
+ """Split one genuinely overlong chunk and transcribe both halves once."""
386
+ audio = AudioSegment.from_file(audio_file)
387
+ midpoint = len(audio) // 2
388
+ ranges = (
389
+ (0, min(len(audio), midpoint + AUDIO_ADAPTIVE_SPLIT_OVERLAP_MS)),
390
+ (max(0, midpoint - AUDIO_ADAPTIVE_SPLIT_OVERLAP_MS), len(audio)),
391
+ )
392
+ split_paths = []
393
+ split_results = []
394
+
395
+ try:
396
+ for start_ms, end_ms in ranges:
397
+ fd, split_path = tempfile.mkstemp(
398
+ prefix="adaptive-audio-split-",
399
+ suffix=".wav",
400
+ dir=self.temp_dir,
401
+ )
402
+ os.close(fd)
403
+ split_paths.append(split_path)
404
+ (
405
+ audio[start_ms:end_ms]
406
+ .set_frame_rate(16000)
407
+ .set_channels(1)
408
+ .set_sample_width(2)
409
+ .export(split_path, format="wav", codec="pcm_s16le")
410
+ .close()
411
+ )
412
+
413
+ split_converter = AudioToTextConverter(
414
+ transcription_model=self.transcription_model,
415
+ transcription_model_provider=self.transcription_model_provider,
416
+ k=self.k,
417
+ min_matches=self.min_matches,
418
+ markdown_output=self.markdown_output,
419
+ llm_api_key=self.llm_api_key,
420
+ max_llm_tokens=self.max_llm_tokens,
421
+ max_output_tokens=self.max_output_tokens,
422
+ temp_dir=self.temp_dir,
423
+ bitrate_quality=self.bitrate_quality,
424
+ timeout_minutes=self.timeout_minutes,
425
+ fallback_stage=self.fallback_stage,
426
+ adaptive_split_depth=self.adaptive_split_depth + 1,
427
+ long_audio_protections_enabled=self.long_audio_protections_enabled,
428
+ prompt_variant=self.prompt_variant,
429
+ is_output_audio_raw=self.is_output_audio_raw,
430
+ )
431
+ split_results.append(
432
+ split_converter.transcribe_audio(split_path, temperature=temperature)
433
+ )
434
+
435
+ merged_transcript = TextMerger(k=self.k, min_matches=self.min_matches).merge_texts(
436
+ split_results[0]["transcript"],
437
+ split_results[1]["transcript"],
438
+ )
439
+ return {
440
+ "transcript": merged_transcript,
441
+ "completion_tokens": sum(item["completion_tokens"] for item in split_results),
442
+ "prompt_tokens": sum(item["prompt_tokens"] for item in split_results),
443
+ "completion_model": self.transcription_model,
444
+ "completion_model_provider": self.transcription_model_provider,
445
+ "finish_reason": "ADAPTIVE_SPLIT",
446
+ "max_output_tokens": self.max_output_tokens,
447
+ "temperature": temperature,
448
+ "prompt_variant": self.prompt_variant,
449
+ "adaptive_split": True,
450
+ "split_results": split_results,
451
+ }
452
+ finally:
453
+ for split_path in split_paths:
454
+ if os.path.exists(split_path):
455
+ os.remove(split_path)
456
+
356
457
  def build_config(self, output_budget: int, temperature: float = 0.0) -> types.GenerateContentConfig:
357
458
  return types.GenerateContentConfig(
358
459
  temperature=temperature,
359
- thinking_config=types.ThinkingConfig(thinking_budget=0),
460
+ thinking_config=types.ThinkingConfig(thinking_level="minimal"),
360
461
  max_output_tokens=output_budget,
361
462
  system_instruction=INJECTION_GUARD_SYSTEM_INSTRUCTION,
362
463
  tools=[],
@@ -433,6 +534,7 @@ class AudioToTextConverter:
433
534
  except ValueError:
434
535
  logger.exception("Unsupported audio format for %s", audio_file)
435
536
  raise
537
+ mime_type = GEMINI_AUDIO_MIME_ALIASES.get(mime_type, mime_type)
436
538
 
437
539
  return client.models.generate_content(
438
540
  model=self.transcription_model,
@@ -455,7 +557,6 @@ class AudioToTextConverter:
455
557
  google_exceptions.ServiceUnavailable,
456
558
  google_exceptions.InternalServerError,
457
559
  genai_errors.ServerError,
458
- genai_errors.APIError,
459
560
  ),
460
561
  tries=8,
461
562
  delay=1,
@@ -518,6 +619,9 @@ class AudioToTextConverter:
518
619
  tail_lines=AUDIO_TAIL_REPETITION_LINES,
519
620
  threshold=AUDIO_TAIL_REPETITION_THRESHOLD,
520
621
  )
622
+ has_repetitive_word_loop = has_excessive_consecutive_word_repetition(
623
+ response_text,
624
+ )
521
625
  usage_metadata = getattr(response, "usage_metadata", None)
522
626
  completion_tokens = getattr(usage_metadata, "candidates_token_count", 0) or 0
523
627
  prompt_tokens = getattr(usage_metadata, "prompt_token_count", 0) or 0
@@ -532,19 +636,84 @@ class AudioToTextConverter:
532
636
  )
533
637
 
534
638
  if finish_reason and "MAX_TOKENS" in finish_reason:
639
+ if (
640
+ self.long_audio_protections_enabled
641
+ and self.adaptive_split_depth < AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH
642
+ ):
643
+ logger.info(
644
+ "Splitting long-audio chunk before fallback after MAX_TOKENS response: %s",
645
+ audio_file,
646
+ )
647
+ return self.transcribe_audio_halves(audio_file, temperature=temperature)
648
+ if has_repetitive_tail or has_repetitive_word_loop:
649
+ raise EmptyDocument(
650
+ message=f"Transcript discarded because repetitive output reached max tokens for audio: {audio_file}",
651
+ code=997,
652
+ )
653
+ if self.adaptive_split_depth < AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH:
654
+ logger.info(
655
+ "Splitting audio chunk after non-repetitive MAX_TOKENS response: %s",
656
+ audio_file,
657
+ )
658
+ return self.transcribe_audio_halves(audio_file, temperature=temperature)
535
659
  raise EmptyDocument(
536
660
  message=f"Transcript truncated because max output tokens were reached for audio: {audio_file}",
537
661
  code=999,
538
662
  )
539
663
 
540
664
  if has_repetitive_tail:
665
+ if (
666
+ self.long_audio_protections_enabled
667
+ and self.adaptive_split_depth < AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH
668
+ ):
669
+ logger.info(
670
+ "Splitting long-audio chunk before fallback after repetitive tail: %s",
671
+ audio_file,
672
+ )
673
+ return self.transcribe_audio_halves(audio_file, temperature=temperature)
541
674
  raise EmptyDocument(
542
675
  message=f"Transcript discarded because repetitive tail was detected for audio: {audio_file}",
543
676
  code=997,
544
677
  )
545
678
 
679
+ if has_repetitive_word_loop:
680
+ if (
681
+ self.long_audio_protections_enabled
682
+ and self.adaptive_split_depth < AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH
683
+ ):
684
+ logger.info(
685
+ "Splitting long-audio chunk before fallback after repetitive word loop: %s",
686
+ audio_file,
687
+ )
688
+ return self.transcribe_audio_halves(audio_file, temperature=temperature)
689
+ raise EmptyDocument(
690
+ message=f"Transcript discarded because repetitive word loop was detected for audio: {audio_file}",
691
+ code=997,
692
+ )
693
+
546
694
  response_text, marker_only = normalize_no_human_speech_marker(response_text)
547
695
 
696
+ is_insignificant_stop = (
697
+ finish_reason
698
+ and "STOP" in finish_reason
699
+ and (
700
+ not response_text.strip()
701
+ or (
702
+ completion_tokens <= 4
703
+ and len(response_text.split()) <= 2
704
+ )
705
+ )
706
+ )
707
+ if (
708
+ self.long_audio_protections_enabled
709
+ and not marker_only
710
+ and is_insignificant_stop
711
+ ):
712
+ raise EmptyDocument(
713
+ message=f"Transcript discarded because STOP returned empty or insignificant output for audio: {audio_file}",
714
+ code=998,
715
+ )
716
+
548
717
  response_dict = {
549
718
  "transcript": "" if marker_only else response_text,
550
719
  "completion_tokens": completion_tokens,
@@ -562,6 +731,24 @@ class AudioToTextConverter:
562
731
  )
563
732
  return response_dict
564
733
  except EmptyDocument as e:
734
+ if (
735
+ e.code == 999
736
+ and self.adaptive_split_depth >= AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH
737
+ and self.fallback_stage == 0
738
+ and self.transcription_model != self.fallback_model
739
+ ):
740
+ return self.run_fallback(
741
+ audio_file=audio_file,
742
+ reason=e.message,
743
+ fallback_model=self.fallback_model,
744
+ fallback_temperature=self.fallback_temperature,
745
+ fallback_stage=2 if self.markdown_output else 1,
746
+ prompt_variant=(
747
+ AUDIO_PROMPT_VARIANT_NON_LITERAL_FALLBACK
748
+ if self.markdown_output
749
+ else self.prompt_variant
750
+ ),
751
+ )
565
752
  if self.should_prompt_fallback_retry(e):
566
753
  return self.run_fallback(
567
754
  audio_file=audio_file,
@@ -654,6 +841,13 @@ class AudioToTextConverter:
654
841
  # Create chunker and extract chunks
655
842
  logger.info("Creating AudioChunker instance...")
656
843
  chunker = AudioChunker(used_file, max_llm_tokens=self.max_llm_tokens)
844
+ self.set_long_audio_protections(chunker.duration_ms)
845
+ logger.info(
846
+ "Long-audio transcription protections enabled: %s (duration: %sms, threshold: %sms)",
847
+ self.long_audio_protections_enabled,
848
+ chunker.duration_ms,
849
+ AUDIO_LONG_DURATION_THRESHOLD_MS,
850
+ )
657
851
  chunks = chunker.extract_chunks()
658
852
 
659
853
  logger.info(f"chunks: {chunks}")
@@ -9,8 +9,6 @@ from google.api_core import exceptions as google_exceptions
9
9
  from retry import retry
10
10
  from concurrent.futures import ThreadPoolExecutor, as_completed
11
11
 
12
- from polytext.processor.transcript_chunker import TranscriptChunker
13
- from polytext.processor.text_merger import TextMerger
14
12
  from polytext.prompts.beautiful_text import BEAUTIFUL_TEXT_PROMPT
15
13
 
16
14
  logger = logging.getLogger(__name__)
@@ -26,6 +24,7 @@ class BeautifulTextConverter:
26
24
  prompt_overhead: int = 1800,
27
25
  tokens_per_char: float = 0.25,
28
26
  overlap_chars: int = 800,
27
+ max_target_chars: int = 12000,
29
28
  ) -> None:
30
29
  self.llm_api_key = llm_api_key
31
30
  self.model = model
@@ -34,19 +33,58 @@ class BeautifulTextConverter:
34
33
  self.prompt_overhead = prompt_overhead
35
34
  self.tokens_per_char = tokens_per_char
36
35
  self.overlap_chars = overlap_chars
36
+ self.max_target_chars = max_target_chars
37
37
 
38
38
  def get_client(self):
39
39
  return genai.Client(api_key=self.llm_api_key) if self.llm_api_key else genai.Client()
40
40
 
41
41
  def chunk_raw_text(self, raw_text: str) -> list[dict]:
42
- chunker = TranscriptChunker(
43
- transcript=raw_text,
44
- max_llm_tokens=self.max_llm_tokens,
45
- prompt_overhead=self.prompt_overhead,
46
- tokens_per_char=self.tokens_per_char,
47
- overlap_chars=self.overlap_chars,
48
- )
49
- return chunker.chunk_transcript()
42
+ text = (raw_text or "").strip()
43
+ if not text:
44
+ return []
45
+
46
+ chunks = []
47
+ start = 0
48
+ index = 0
49
+
50
+ while start < len(text):
51
+ proposed_end = min(start + self.max_target_chars, len(text))
52
+ end = self._find_chunk_end(text, start, proposed_end)
53
+ target_text = text[start:end].strip()
54
+
55
+ if not target_text:
56
+ break
57
+
58
+ chunks.append(
59
+ {
60
+ "index": index,
61
+ "target_text": target_text,
62
+ }
63
+ )
64
+ start = end
65
+ index += 1
66
+
67
+ return chunks
68
+
69
+ def _find_chunk_end(self, text: str, start: int, proposed_end: int) -> int:
70
+ if proposed_end >= len(text):
71
+ return len(text)
72
+
73
+ minimum_end = start + int(self.max_target_chars * 0.65)
74
+ chunk_window = text[minimum_end:proposed_end]
75
+ sentence_boundaries = list(re.finditer(r"(?<=[.!?])\s+", chunk_window))
76
+ if sentence_boundaries:
77
+ return minimum_end + sentence_boundaries[-1].end()
78
+
79
+ paragraph_boundary = text.rfind("\n\n", minimum_end, proposed_end)
80
+ if paragraph_boundary != -1:
81
+ return paragraph_boundary + 2
82
+
83
+ whitespace_boundary = text.rfind(" ", minimum_end, proposed_end)
84
+ if whitespace_boundary != -1:
85
+ return whitespace_boundary + 1
86
+
87
+ return proposed_end
50
88
 
51
89
  @retry(
52
90
  (
@@ -60,11 +98,18 @@ class BeautifulTextConverter:
60
98
  backoff=2,
61
99
  logger=logger,
62
100
  )
63
- def process_chunk(self, client, chunk_text: str, index: int) -> dict:
101
+ def process_chunk(self, client, chunk: dict, index: int) -> dict:
64
102
  logger.info("Processing beautiful text chunk %s", index + 1)
65
103
  start_time = time.time()
66
104
 
105
+ target_text = chunk["target_text"]
106
+
67
107
  config = types.GenerateContentConfig(
108
+ temperature=0,
109
+ thinking_config=types.ThinkingConfig(thinking_budget=0),
110
+ max_output_tokens=self.max_llm_tokens,
111
+ tools=[],
112
+ automatic_function_calling=types.AutomaticFunctionCallingConfig(disable=True),
68
113
  safety_settings=[
69
114
  types.SafetySetting(
70
115
  category=types.HarmCategory.HARM_CATEGORY_HATE_SPEECH,
@@ -87,7 +132,11 @@ class BeautifulTextConverter:
87
132
 
88
133
  response = client.models.generate_content(
89
134
  model=self.model,
90
- contents=[BEAUTIFUL_TEXT_PROMPT, chunk_text],
135
+ contents=[
136
+ BEAUTIFUL_TEXT_PROMPT,
137
+ "TARGET TEXT TO CLEAN",
138
+ target_text,
139
+ ],
91
140
  config=config,
92
141
  )
93
142
 
@@ -100,7 +149,7 @@ class BeautifulTextConverter:
100
149
  }
101
150
 
102
151
  def merge_cleaned_chunks(self, chunks: list[str]) -> str:
103
- return TextMerger(llm_api_key=self.llm_api_key).merge_chunks(chunks=chunks)
152
+ return "\n\n".join(chunk.strip() for chunk in chunks if chunk.strip())
104
153
 
105
154
  def _convert_markdown_to_json(self, markdown_text: str) -> dict:
106
155
  if not markdown_text.strip():
@@ -181,7 +230,7 @@ class BeautifulTextConverter:
181
230
 
182
231
  with ThreadPoolExecutor() as executor:
183
232
  future_to_index = {
184
- executor.submit(self.process_chunk, client, chunk["text"], chunk["index"]): chunk["index"]
233
+ executor.submit(self.process_chunk, client, chunk, chunk["index"]): chunk["index"]
185
234
  for chunk in chunks
186
235
  }
187
236
 
@@ -242,7 +242,9 @@ class DocumentOCRToTextConverter:
242
242
  return error.code in OCR_RETRIABLE_OUTPUT_ERROR_CODES
243
243
 
244
244
  def should_fallback_temperature_retry(self, error: EmptyDocument, temperature: float) -> bool:
245
- expected_stage = 1 if self.markdown_output else 0
245
+ # The non-literal prompt retry advances every OCR mode to stage 1.
246
+ # Plain-text document OCR must therefore also try the fallback model at stage 1.
247
+ expected_stage = 1
246
248
  if self.fallback_stage != expected_stage:
247
249
  return False
248
250
  if error.code not in OCR_RETRIABLE_OUTPUT_ERROR_CODES:
@@ -368,6 +370,8 @@ class DocumentOCRToTextConverter:
368
370
  config = types.GenerateContentConfig(
369
371
  temperature=temperature,
370
372
  max_output_tokens=self.max_output_tokens,
373
+ tools=[],
374
+ automatic_function_calling=types.AutomaticFunctionCallingConfig(disable=True),
371
375
  safety_settings=[
372
376
  types.SafetySetting(
373
377
  category=types.HarmCategory.HARM_CATEGORY_HATE_SPEECH,
@@ -37,6 +37,14 @@ def has_consecutive_repetition(items: list[str], min_run_length: int = 3) -> boo
37
37
  return False
38
38
 
39
39
 
40
+ def has_excessive_consecutive_word_repetition(
41
+ text: str,
42
+ min_run_length: int = 12,
43
+ ) -> bool:
44
+ words = re.findall(r"\b\w+\b", (text or "").casefold())
45
+ return has_consecutive_repetition(words, min_run_length=min_run_length)
46
+
47
+
40
48
  def tail_has_excessive_repetition(
41
49
  text: str,
42
50
  tail_lines: int,
@@ -352,6 +352,8 @@ class OCRToTextConverter:
352
352
  config = types.GenerateContentConfig(
353
353
  temperature=temperature,
354
354
  max_output_tokens=self.max_output_tokens,
355
+ tools=[],
356
+ automatic_function_calling=types.AutomaticFunctionCallingConfig(disable=True),
355
357
  safety_settings=[
356
358
  types.SafetySetting(
357
359
  category=types.HarmCategory.HARM_CATEGORY_HATE_SPEECH,
@@ -141,6 +141,8 @@ class TextToMdConverter:
141
141
  start_time = time.time()
142
142
 
143
143
  config = types.GenerateContentConfig(
144
+ tools=[],
145
+ automatic_function_calling=types.AutomaticFunctionCallingConfig(disable=True),
144
146
  safety_settings=[
145
147
  types.SafetySetting(category=types.HarmCategory.HARM_CATEGORY_HATE_SPEECH, threshold=types.HarmBlockThreshold.BLOCK_NONE),
146
148
  types.SafetySetting(category=types.HarmCategory.HARM_CATEGORY_DANGEROUS_CONTENT, threshold=types.HarmBlockThreshold.BLOCK_NONE),
@@ -235,4 +237,4 @@ class TextToMdConverter:
235
237
 
236
238
  logging.info(f"**** FINISH ****")
237
239
 
238
- return result_dict
240
+ return result_dict
@@ -12,7 +12,7 @@ def convert_video_to_audio(video_file: str , bitrate_quality: int =9) -> str:
12
12
 
13
13
  Args:
14
14
  video_file (str): Path to the video file.
15
- bitrate_quality (int, optional): Variable bitrate quality from 0-9 (9 being lowest). Defaults to 9.
15
+ bitrate_quality (int, optional): Retained for backward compatibility; WAV output is lossless.
16
16
 
17
17
  Returns:
18
18
  str: Path to the converted audio file.
@@ -22,12 +22,12 @@ def convert_video_to_audio(video_file: str , bitrate_quality: int =9) -> str:
22
22
  Exception: If any other error occurs during conversion
23
23
  """
24
24
 
25
- logger.info(f"Converting video to audio with bitrate quality {bitrate_quality}.")
25
+ logger.info("Converting video to lossless 16 kHz mono WAV.")
26
26
 
27
27
  temp_audio_path = None
28
28
  try:
29
29
  # Create temporary file for audio output
30
- fd, temp_audio_path = tempfile.mkstemp(suffix='.mp3')
30
+ fd, temp_audio_path = tempfile.mkstemp(suffix='.wav')
31
31
  os.close(fd)
32
32
 
33
33
  # Simple efficient pipeline
@@ -35,9 +35,7 @@ def convert_video_to_audio(video_file: str , bitrate_quality: int =9) -> str:
35
35
  ffmpeg
36
36
  .input(video_file)
37
37
  .output(temp_audio_path,
38
- acodec='libmp3lame',
39
- # ab='64k',
40
- q=bitrate_quality, # Variable bitrate quality (0-9, 9 being lowest)
38
+ acodec='pcm_s16le',
41
39
  ac=1, # Convert to mono
42
40
  ar=16000, # Lower sample rate
43
41
  vn=None,
@@ -458,7 +458,6 @@ class BaseLoader:
458
458
  ocr_provider=self.provider,
459
459
  ocr_model=self.ocr_model,
460
460
  include_image_descriptions=self.include_image_descriptions,
461
- allow_partial_ocr_failures=True,
462
461
  **document_kwargs,
463
462
  )
464
463