polytext 0.2.8b2__tar.gz → 0.2.8b4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. {polytext-0.2.8b2 → polytext-0.2.8b4}/PKG-INFO +2 -2
  2. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/audio_to_text.py +222 -23
  3. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/beautiful_text.py +63 -14
  4. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/document_ocr_to_text.py +5 -1
  5. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/gemini_quality_guards.py +8 -0
  6. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/ocr_to_text.py +2 -0
  7. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/text_to_md.py +3 -1
  8. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/video_to_audio.py +4 -6
  9. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/base.py +0 -1
  10. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/youtube_llm.py +2 -0
  11. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/processor/audio_chunker.py +4 -5
  12. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/processor/text_merger.py +56 -30
  13. polytext-0.2.8b4/polytext/prompts/beautiful_text.py +97 -0
  14. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/prompts/transcription.py +32 -12
  15. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext.egg-info/PKG-INFO +2 -2
  16. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext.egg-info/requires.txt +1 -1
  17. {polytext-0.2.8b2 → polytext-0.2.8b4}/setup.py +1 -1
  18. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_audio_chunker.py +8 -2
  19. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_audio_transcription_model_migration.py +333 -22
  20. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_ocr_fallbacks.py +32 -0
  21. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_ocr_image_descriptions.py +13 -0
  22. polytext-0.2.8b2/polytext/prompts/beautiful_text.py +0 -61
  23. {polytext-0.2.8b2 → polytext-0.2.8b4}/LICENSE +0 -0
  24. {polytext-0.2.8b2 → polytext-0.2.8b4}/README.md +0 -0
  25. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/__init__.py +0 -0
  26. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/__init__.py +0 -0
  27. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/base.py +0 -0
  28. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/document_ocr_to_text_azure_oai.py +0 -0
  29. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/html_to_md.py +0 -0
  30. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/md_to_text.py +0 -0
  31. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/ocr_to_text_azure_oai.py +0 -0
  32. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/pdf.py +0 -0
  33. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/exceptions/__init__.py +0 -0
  34. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/exceptions/base.py +0 -0
  35. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/generator/__init__.py +0 -0
  36. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/generator/pdf.py +0 -0
  37. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/__init__.py +0 -0
  38. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/audio.py +0 -0
  39. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/aws_auth.py +0 -0
  40. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/document.py +0 -0
  41. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/document_ocr.py +0 -0
  42. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/downloader/__init__.py +0 -0
  43. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/downloader/downloader.py +0 -0
  44. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/html.py +0 -0
  45. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/markdown.py +0 -0
  46. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/notebook.py +0 -0
  47. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/ocr.py +0 -0
  48. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/plain_text.py +0 -0
  49. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/video.py +0 -0
  50. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/xml_xbrl.py +0 -0
  51. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/youtube.py +0 -0
  52. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/processor/__init__.py +0 -0
  53. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/processor/transcript_chunker.py +0 -0
  54. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/prompts/__init__.py +0 -0
  55. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/prompts/ocr.py +0 -0
  56. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/prompts/text_merging.py +0 -0
  57. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/prompts/text_to_md.py +0 -0
  58. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/utils/__init__.py +0 -0
  59. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/utils/utils.py +0 -0
  60. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext.egg-info/SOURCES.txt +0 -0
  61. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext.egg-info/dependency_links.txt +0 -0
  62. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext.egg-info/not-zip-safe +0 -0
  63. {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext.egg-info/top_level.txt +0 -0
  64. {polytext-0.2.8b2 → polytext-0.2.8b4}/pyproject.toml +0 -0
  65. {polytext-0.2.8b2 → polytext-0.2.8b4}/setup.cfg +0 -0
  66. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_audio_comparison_helpers.py +0 -0
  67. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_aws_auth.py +0 -0
  68. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_base_loader_error_mapping.py +0 -0
  69. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_beautiful_text_manual.py +0 -0
  70. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_compare_audio_models.py +0 -0
  71. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_compare_document_ocr_to_text_models.py +0 -0
  72. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_compare_ocr_to_text_models.py +0 -0
  73. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_compare_youtube_models.py +0 -0
  74. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_dowload_audio_from_youtube.py +0 -0
  75. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_dowload_audio_from_youtube_helpers.py +0 -0
  76. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_extracted_text_whitespace.py +0 -0
  77. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_gemini_quality_guards.py +0 -0
  78. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_audio_transcript_from_gcs.py +0 -0
  79. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_customized_pdf_from_markdown.py +0 -0
  80. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_document_ocr.py +0 -0
  81. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_document_ocr_azure_oai.py +0 -0
  82. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_document_text.py +0 -0
  83. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_document_text_from_gcs.py +0 -0
  84. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_ocr_from_image.py +0 -0
  85. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_text_from_markdown.py +0 -0
  86. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_video_transcript_from_gcs.py +0 -0
  87. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_library.py +0 -0
  88. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_markdown_loader_gzip.py +0 -0
  89. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_markitdown_html.py +0 -0
  90. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_notebook_loader.py +0 -0
  91. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_pain_text.py +0 -0
  92. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_pdf_conversion_error.py +0 -0
  93. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_python_version_metadata.py +0 -0
  94. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_split_audio_with_llm.py +0 -0
  95. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_transcribe_s3_images_from_csv.py +0 -0
  96. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_transcribe_s3_images_from_csv_script.py +0 -0
  97. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_xml_xbrl_loader.py +0 -0
  98. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_youtube_gemini_minimal_check.py +0 -0
  99. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_youtube_llm_fallbacks.py +0 -0
  100. {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_youtube_transcript.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: polytext
3
- Version: 0.2.8b2
3
+ Version: 0.2.8b4
4
4
  Summary: Python utilities to simplify document files management
5
5
  Home-page: https://github.com/docsity/polytext
6
6
  Author: Matteo Senardi
@@ -25,7 +25,7 @@ Requires-Dist: markdown-to-json==2.1.2
25
25
  Requires-Dist: python-docx==1.1.2
26
26
  Requires-Dist: google-api-core>=2.24.2
27
27
  Requires-Dist: google-cloud-storage<3.0.0,>=2.17
28
- Requires-Dist: google-genai>=1.16.1
28
+ Requires-Dist: google-genai==2.22.0
29
29
  Requires-Dist: openai==2.26.0
30
30
  Requires-Dist: boto3>=1.42.64
31
31
  Requires-Dist: botocore>=1.42.64
@@ -14,6 +14,7 @@ from google.genai import types
14
14
  from google.genai import errors as genai_errors
15
15
  from concurrent.futures import ThreadPoolExecutor, as_completed
16
16
  from google.api_core import exceptions as google_exceptions
17
+ from pydub import AudioSegment
17
18
 
18
19
  from ..exceptions import EmptyDocument
19
20
  from ..prompts.transcription import (
@@ -25,13 +26,22 @@ from ..prompts.transcription import (
25
26
  )
26
27
  from ..processor.audio_chunker import AudioChunker
27
28
  from ..processor.text_merger import TextMerger
28
- from .gemini_quality_guards import extract_finish_reason, tail_has_excessive_repetition
29
+ from .gemini_quality_guards import (
30
+ extract_finish_reason,
31
+ has_excessive_consecutive_word_repetition,
32
+ tail_has_excessive_repetition,
33
+ )
29
34
 
30
35
  logger = logging.getLogger(__name__)
31
36
 
32
37
  SUPPORTED_MIME_TYPES = {
33
38
  'audio/x-aac', 'audio/flac', 'audio/mp3', 'audio/m4a', 'audio/mpeg',
34
- 'audio/mpga', 'audio/mp4', 'audio/opus', 'audio/pcm', 'audio/wav', 'audio/webm'
39
+ 'audio/mpga', 'audio/mp4', 'audio/opus', 'audio/pcm', 'audio/wav', 'audio/x-wav', 'audio/webm'
40
+ }
41
+
42
+ GEMINI_AUDIO_MIME_ALIASES = {
43
+ 'audio/x-aac': 'audio/aac',
44
+ 'audio/x-wav': 'audio/wav',
35
45
  }
36
46
 
37
47
  INJECTION_GUARD_SYSTEM_INSTRUCTION = (
@@ -48,16 +58,20 @@ INJECTION_GUARD_SYSTEM_INSTRUCTION = (
48
58
  )
49
59
 
50
60
  AUDIO_MIN_OUTPUT_TOKENS = 500
61
+ AUDIO_DEFAULT_MAX_OUTPUT_TOKENS = 4096
62
+ AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH = 1
63
+ AUDIO_ADAPTIVE_SPLIT_OVERLAP_MS = 2000
64
+ AUDIO_LONG_DURATION_THRESHOLD_MS = 80 * 60 * 1000
51
65
  AUDIO_TAIL_REPETITION_LINES = int(os.getenv("AUDIO_TAIL_REPETITION_LINES", "200"))
52
66
  AUDIO_TAIL_REPETITION_THRESHOLD = float(os.getenv("AUDIO_TAIL_REPETITION_THRESHOLD", "0.35"))
53
67
  AUDIO_FALLBACK_SOURCE_PATTERN = os.getenv("AUDIO_FALLBACK_SOURCE_PATTERN", "flash-lite")
54
- AUDIO_FALLBACK_MODEL = os.getenv("AUDIO_FALLBACK_MODEL", "gemini-3-flash-preview")
68
+ AUDIO_FALLBACK_MODEL = os.getenv("AUDIO_FALLBACK_MODEL", "gemini-3.5-flash-lite")
55
69
  AUDIO_FALLBACK_TEMPERATURE = float(os.getenv("AUDIO_FALLBACK_TEMPERATURE", "1.0"))
56
70
  AUDIO_FINAL_FALLBACK_MODEL = os.getenv("AUDIO_FINAL_FALLBACK_MODEL", "gemini-3.5-flash")
57
71
  AUDIO_FILE_UPLOAD_THRESHOLD_BYTES = 20 * 1024 * 1024
58
72
  AUDIO_PROMPT_VARIANT_DEFAULT = "default"
59
73
  AUDIO_PROMPT_VARIANT_NON_LITERAL_FALLBACK = "non_literal_fallback"
60
- AUDIO_RETRIABLE_OUTPUT_ERROR_CODES = (996, 997, 999)
74
+ AUDIO_RETRIABLE_OUTPUT_ERROR_CODES = (996, 997, 998, 999)
61
75
  NO_HUMAN_SPEECH_MARKER = "no human speech detected"
62
76
 
63
77
 
@@ -80,23 +94,30 @@ def add_line_break_after_each_sentence(text: str) -> str:
80
94
  if not text:
81
95
  return text
82
96
 
97
+ line_feed = chr(10)
83
98
  lines = text.splitlines()
84
99
  formatted_lines = []
85
100
 
86
101
  for line in lines:
87
102
  stripped_line = line.strip()
103
+
88
104
  if not stripped_line:
89
105
  formatted_lines.append("")
90
106
  continue
107
+
91
108
  if re.match(r"^#{1,6}\s+", stripped_line):
92
109
  formatted_lines.append(stripped_line)
93
110
  continue
94
111
 
95
112
  normalized_line = re.sub(r"\s+", " ", stripped_line)
96
- normalized_line = re.sub(r"([.!?])\s+", r"\1\\n ", normalized_line)
113
+ normalized_line = re.sub(
114
+ r"([.!?])\s+",
115
+ lambda match: f"{match.group(1)}{line_feed} ",
116
+ normalized_line,
117
+ )
97
118
  formatted_lines.append(normalized_line)
98
119
 
99
- return "\\n ".join(formatted_lines).strip()
120
+ return line_feed.join(formatted_lines).strip()
100
121
 
101
122
 
102
123
  def create_ascii_safe_upload_copy(audio_file: str) -> tuple[str, str | None]:
@@ -116,33 +137,32 @@ def create_ascii_safe_upload_copy(audio_file: str) -> tuple[str, str | None]:
116
137
 
117
138
  def compress_and_convert_audio(input_path: str, bitrate_quality: int = 9) -> str:
118
139
  """
119
- Compress and convert an audio file to MP3 using ffmpeg.
140
+ Normalize an audio file to lossless 16 kHz mono WAV using ffmpeg.
120
141
 
121
142
  Args:
122
143
  input_path (str): Path to the original audio file
123
- bitrate_quality (int, optional): Variable bitrate quality from 0-9 (9 being lowest). Defaults to 9
144
+ bitrate_quality (int, optional): Retained for backward compatibility; WAV output is lossless.
124
145
 
125
146
  Returns:
126
- str: Path to the temporary compressed/converted MP3 file
147
+ str: Path to the temporary normalized WAV file
127
148
 
128
149
  Raises:
129
150
  RuntimeError: If FFmpeg compression/conversion fails
130
151
 
131
152
  Notes:
132
- - Creates a temporary MP3 file that should be deleted after use
133
- - Converts audio to mono and 16kHz sample rate for smaller file size
153
+ - Creates a temporary WAV file that should be deleted after use
154
+ - Converts audio to 16-bit PCM mono at 16kHz
134
155
  - Uses maximum available CPU threads for faster processing
135
156
  """
136
157
  # Create temporary file for audio output
137
- fd, temp_audio_path = tempfile.mkstemp(suffix='.mp3')
158
+ fd, temp_audio_path = tempfile.mkstemp(suffix='.wav')
138
159
  os.close(fd)
139
160
 
140
- logger.info(f"Compressing audio to bitrate quality: {bitrate_quality}")
161
+ logger.info("Normalizing audio to lossless 16 kHz mono WAV")
141
162
  try:
142
163
  ffmpeg.input(input_path).output(
143
164
  temp_audio_path,
144
- q=bitrate_quality, # Variable bitrate quality (0-9, 9 being lowest)
145
- acodec='libmp3lame',
165
+ acodec='pcm_s16le',
146
166
  ac=1, # Convert to mono
147
167
  ar=16000, # Lower sample rate
148
168
  vn=None,
@@ -155,7 +175,7 @@ def compress_and_convert_audio(input_path: str, bitrate_quality: int = 9) -> str
155
175
  os.unlink(temp_audio_path)
156
176
  raise
157
177
 
158
- logger.info(f"Successfully converted and compressed audio: {temp_audio_path}")
178
+ logger.info(f"Successfully normalized audio: {temp_audio_path}")
159
179
  return temp_audio_path
160
180
 
161
181
 
@@ -182,7 +202,7 @@ def transcribe_full_audio(audio_file, markdown_output: bool = False,
182
202
  bitrate_quality (int, optional): Variable bitrate quality from 0-9 (9 being lowest). Defaults to 9
183
203
  timeout_minutes (int, optional): Number of minutes to wait for a response. Defaults to None.
184
204
  max_llm_tokens (int, optional): Token budget used for audio chunk sizing. Defaults to 4250.
185
- max_output_tokens (int | None, optional): Maximum Gemini output tokens. Defaults to `max_llm_tokens`.
205
+ max_output_tokens (int | None, optional): Maximum Gemini output tokens. Defaults to 4096.
186
206
  is_output_audio_raw (bool, optional): If True, use the raw Markdown audio prompt.
187
207
  If False, use the formatted Markdown audio prompt. Defaults to True.
188
208
 
@@ -204,6 +224,8 @@ class AudioToTextConverter:
204
224
  max_output_tokens: int | None = None, temp_dir: str = "temp",
205
225
  bitrate_quality: int = 9, timeout_minutes: int = None,
206
226
  fallback_stage: int = 0,
227
+ adaptive_split_depth: int = 0,
228
+ long_audio_protections_enabled: bool = False,
207
229
  prompt_variant: str = AUDIO_PROMPT_VARIANT_DEFAULT,
208
230
  is_output_audio_raw: bool = True):
209
231
  """
@@ -218,12 +240,14 @@ class AudioToTextConverter:
218
240
  llm_api_key (str, optional): Override API key for language model. Defaults to None.
219
241
  max_llm_tokens (int): Token budget used to size audio chunks. Defaults to 4250.
220
242
  max_output_tokens (int | None): Maximum number of output tokens for Gemini generation.
221
- Defaults to `max_llm_tokens`.
243
+ Defaults to 4096.
222
244
  temp_dir (str): Directory for temporary files. Defaults to "temp".
223
245
  bitrate_quality (int, optional): Variable bitrate quality from 0-9 (9 being lowest). Defaults to 9
224
246
  timeout_minutes (int): Number of minutes to wait for a response.
225
247
  fallback_stage (int, optional): Internal retry stage used by fallback attempts.
226
248
  Defaults to 0.
249
+ long_audio_protections_enabled (bool, optional): Apply stricter recovery rules inherited
250
+ from an original audio longer than 80 minutes. Defaults to False.
227
251
  prompt_variant (str, optional): Prompt variant used by this attempt.
228
252
  Defaults to "default".
229
253
  is_output_audio_raw (bool, optional): If True, use the raw Markdown audio prompt.
@@ -240,12 +264,18 @@ class AudioToTextConverter:
240
264
  self.markdown_output = markdown_output
241
265
  self.llm_api_key = llm_api_key
242
266
  self.max_llm_tokens = max(max_llm_tokens, AUDIO_MIN_OUTPUT_TOKENS)
243
- requested_output_tokens = self.max_llm_tokens if max_output_tokens is None else max_output_tokens
267
+ requested_output_tokens = (
268
+ AUDIO_DEFAULT_MAX_OUTPUT_TOKENS
269
+ if max_output_tokens is None
270
+ else max_output_tokens
271
+ )
244
272
  self.max_output_tokens = max(requested_output_tokens, AUDIO_MIN_OUTPUT_TOKENS)
245
273
  self.chunked_audio = False
246
274
  self.bitrate_quality = bitrate_quality
247
275
  self.timeout_minutes = timeout_minutes
248
276
  self.fallback_stage = fallback_stage
277
+ self.adaptive_split_depth = adaptive_split_depth
278
+ self.long_audio_protections_enabled = long_audio_protections_enabled
249
279
  self.prompt_variant = prompt_variant
250
280
  self.fallback_source_pattern = AUDIO_FALLBACK_SOURCE_PATTERN
251
281
  self.fallback_model = AUDIO_FALLBACK_MODEL
@@ -269,6 +299,9 @@ class AudioToTextConverter:
269
299
  return AUDIO_TO_MARKDOWN_PROMPT
270
300
  return AUDIO_TO_PLAIN_TEXT_PROMPT
271
301
 
302
+ def set_long_audio_protections(self, duration_ms: int) -> None:
303
+ self.long_audio_protections_enabled = duration_ms > AUDIO_LONG_DURATION_THRESHOLD_MS
304
+
272
305
  def should_prompt_fallback_retry(self, error: EmptyDocument) -> bool:
273
306
  if self.fallback_stage != 0:
274
307
  return False
@@ -331,6 +364,8 @@ class AudioToTextConverter:
331
364
  bitrate_quality=self.bitrate_quality,
332
365
  timeout_minutes=self.timeout_minutes,
333
366
  fallback_stage=fallback_stage,
367
+ adaptive_split_depth=self.adaptive_split_depth,
368
+ long_audio_protections_enabled=self.long_audio_protections_enabled,
334
369
  prompt_variant=resolved_prompt_variant,
335
370
  is_output_audio_raw=self.is_output_audio_raw,
336
371
  )
@@ -346,10 +381,83 @@ class AudioToTextConverter:
346
381
  result.setdefault("fallback_to_prompt_variant", resolved_prompt_variant)
347
382
  return result
348
383
 
384
+ def transcribe_audio_halves(self, audio_file: str, temperature: float = 0.0) -> dict:
385
+ """Split one genuinely overlong chunk and transcribe both halves once."""
386
+ audio = AudioSegment.from_file(audio_file)
387
+ midpoint = len(audio) // 2
388
+ ranges = (
389
+ (0, min(len(audio), midpoint + AUDIO_ADAPTIVE_SPLIT_OVERLAP_MS)),
390
+ (max(0, midpoint - AUDIO_ADAPTIVE_SPLIT_OVERLAP_MS), len(audio)),
391
+ )
392
+ split_paths = []
393
+ split_results = []
394
+
395
+ try:
396
+ for start_ms, end_ms in ranges:
397
+ fd, split_path = tempfile.mkstemp(
398
+ prefix="adaptive-audio-split-",
399
+ suffix=".wav",
400
+ dir=self.temp_dir,
401
+ )
402
+ os.close(fd)
403
+ split_paths.append(split_path)
404
+ (
405
+ audio[start_ms:end_ms]
406
+ .set_frame_rate(16000)
407
+ .set_channels(1)
408
+ .set_sample_width(2)
409
+ .export(split_path, format="wav", codec="pcm_s16le")
410
+ .close()
411
+ )
412
+
413
+ split_converter = AudioToTextConverter(
414
+ transcription_model=self.transcription_model,
415
+ transcription_model_provider=self.transcription_model_provider,
416
+ k=self.k,
417
+ min_matches=self.min_matches,
418
+ markdown_output=self.markdown_output,
419
+ llm_api_key=self.llm_api_key,
420
+ max_llm_tokens=self.max_llm_tokens,
421
+ max_output_tokens=self.max_output_tokens,
422
+ temp_dir=self.temp_dir,
423
+ bitrate_quality=self.bitrate_quality,
424
+ timeout_minutes=self.timeout_minutes,
425
+ fallback_stage=self.fallback_stage,
426
+ adaptive_split_depth=self.adaptive_split_depth + 1,
427
+ long_audio_protections_enabled=self.long_audio_protections_enabled,
428
+ prompt_variant=self.prompt_variant,
429
+ is_output_audio_raw=self.is_output_audio_raw,
430
+ )
431
+ split_results.append(
432
+ split_converter.transcribe_audio(split_path, temperature=temperature)
433
+ )
434
+
435
+ merged_transcript = TextMerger(k=self.k, min_matches=self.min_matches).merge_texts(
436
+ split_results[0]["transcript"],
437
+ split_results[1]["transcript"],
438
+ )
439
+ return {
440
+ "transcript": merged_transcript,
441
+ "completion_tokens": sum(item["completion_tokens"] for item in split_results),
442
+ "prompt_tokens": sum(item["prompt_tokens"] for item in split_results),
443
+ "completion_model": self.transcription_model,
444
+ "completion_model_provider": self.transcription_model_provider,
445
+ "finish_reason": "ADAPTIVE_SPLIT",
446
+ "max_output_tokens": self.max_output_tokens,
447
+ "temperature": temperature,
448
+ "prompt_variant": self.prompt_variant,
449
+ "adaptive_split": True,
450
+ "split_results": split_results,
451
+ }
452
+ finally:
453
+ for split_path in split_paths:
454
+ if os.path.exists(split_path):
455
+ os.remove(split_path)
456
+
349
457
  def build_config(self, output_budget: int, temperature: float = 0.0) -> types.GenerateContentConfig:
350
458
  return types.GenerateContentConfig(
351
459
  temperature=temperature,
352
- thinking_config=types.ThinkingConfig(thinking_budget=0),
460
+ thinking_config=types.ThinkingConfig(thinking_level="minimal"),
353
461
  max_output_tokens=output_budget,
354
462
  system_instruction=INJECTION_GUARD_SYSTEM_INSTRUCTION,
355
463
  tools=[],
@@ -426,6 +534,7 @@ class AudioToTextConverter:
426
534
  except ValueError:
427
535
  logger.exception("Unsupported audio format for %s", audio_file)
428
536
  raise
537
+ mime_type = GEMINI_AUDIO_MIME_ALIASES.get(mime_type, mime_type)
429
538
 
430
539
  return client.models.generate_content(
431
540
  model=self.transcription_model,
@@ -448,7 +557,6 @@ class AudioToTextConverter:
448
557
  google_exceptions.ServiceUnavailable,
449
558
  google_exceptions.InternalServerError,
450
559
  genai_errors.ServerError,
451
- genai_errors.APIError,
452
560
  ),
453
561
  tries=8,
454
562
  delay=1,
@@ -511,6 +619,9 @@ class AudioToTextConverter:
511
619
  tail_lines=AUDIO_TAIL_REPETITION_LINES,
512
620
  threshold=AUDIO_TAIL_REPETITION_THRESHOLD,
513
621
  )
622
+ has_repetitive_word_loop = has_excessive_consecutive_word_repetition(
623
+ response_text,
624
+ )
514
625
  usage_metadata = getattr(response, "usage_metadata", None)
515
626
  completion_tokens = getattr(usage_metadata, "candidates_token_count", 0) or 0
516
627
  prompt_tokens = getattr(usage_metadata, "prompt_token_count", 0) or 0
@@ -525,20 +636,83 @@ class AudioToTextConverter:
525
636
  )
526
637
 
527
638
  if finish_reason and "MAX_TOKENS" in finish_reason:
639
+ if (
640
+ self.long_audio_protections_enabled
641
+ and self.adaptive_split_depth < AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH
642
+ ):
643
+ logger.info(
644
+ "Splitting long-audio chunk before fallback after MAX_TOKENS response: %s",
645
+ audio_file,
646
+ )
647
+ return self.transcribe_audio_halves(audio_file, temperature=temperature)
648
+ if has_repetitive_tail or has_repetitive_word_loop:
649
+ raise EmptyDocument(
650
+ message=f"Transcript discarded because repetitive output reached max tokens for audio: {audio_file}",
651
+ code=997,
652
+ )
653
+ if self.adaptive_split_depth < AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH:
654
+ logger.info(
655
+ "Splitting audio chunk after non-repetitive MAX_TOKENS response: %s",
656
+ audio_file,
657
+ )
658
+ return self.transcribe_audio_halves(audio_file, temperature=temperature)
528
659
  raise EmptyDocument(
529
660
  message=f"Transcript truncated because max output tokens were reached for audio: {audio_file}",
530
661
  code=999,
531
662
  )
532
663
 
533
664
  if has_repetitive_tail:
665
+ if (
666
+ self.long_audio_protections_enabled
667
+ and self.adaptive_split_depth < AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH
668
+ ):
669
+ logger.info(
670
+ "Splitting long-audio chunk before fallback after repetitive tail: %s",
671
+ audio_file,
672
+ )
673
+ return self.transcribe_audio_halves(audio_file, temperature=temperature)
534
674
  raise EmptyDocument(
535
675
  message=f"Transcript discarded because repetitive tail was detected for audio: {audio_file}",
536
676
  code=997,
537
677
  )
538
678
 
679
+ if has_repetitive_word_loop:
680
+ if (
681
+ self.long_audio_protections_enabled
682
+ and self.adaptive_split_depth < AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH
683
+ ):
684
+ logger.info(
685
+ "Splitting long-audio chunk before fallback after repetitive word loop: %s",
686
+ audio_file,
687
+ )
688
+ return self.transcribe_audio_halves(audio_file, temperature=temperature)
689
+ raise EmptyDocument(
690
+ message=f"Transcript discarded because repetitive word loop was detected for audio: {audio_file}",
691
+ code=997,
692
+ )
693
+
539
694
  response_text, marker_only = normalize_no_human_speech_marker(response_text)
540
- if not marker_only:
541
- response_text = self.format_audio_output_text(response_text)
695
+
696
+ is_insignificant_stop = (
697
+ finish_reason
698
+ and "STOP" in finish_reason
699
+ and (
700
+ not response_text.strip()
701
+ or (
702
+ completion_tokens <= 4
703
+ and len(response_text.split()) <= 2
704
+ )
705
+ )
706
+ )
707
+ if (
708
+ self.long_audio_protections_enabled
709
+ and not marker_only
710
+ and is_insignificant_stop
711
+ ):
712
+ raise EmptyDocument(
713
+ message=f"Transcript discarded because STOP returned empty or insignificant output for audio: {audio_file}",
714
+ code=998,
715
+ )
542
716
 
543
717
  response_dict = {
544
718
  "transcript": "" if marker_only else response_text,
@@ -557,6 +731,24 @@ class AudioToTextConverter:
557
731
  )
558
732
  return response_dict
559
733
  except EmptyDocument as e:
734
+ if (
735
+ e.code == 999
736
+ and self.adaptive_split_depth >= AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH
737
+ and self.fallback_stage == 0
738
+ and self.transcription_model != self.fallback_model
739
+ ):
740
+ return self.run_fallback(
741
+ audio_file=audio_file,
742
+ reason=e.message,
743
+ fallback_model=self.fallback_model,
744
+ fallback_temperature=self.fallback_temperature,
745
+ fallback_stage=2 if self.markdown_output else 1,
746
+ prompt_variant=(
747
+ AUDIO_PROMPT_VARIANT_NON_LITERAL_FALLBACK
748
+ if self.markdown_output
749
+ else self.prompt_variant
750
+ ),
751
+ )
560
752
  if self.should_prompt_fallback_retry(e):
561
753
  return self.run_fallback(
562
754
  audio_file=audio_file,
@@ -649,6 +841,13 @@ class AudioToTextConverter:
649
841
  # Create chunker and extract chunks
650
842
  logger.info("Creating AudioChunker instance...")
651
843
  chunker = AudioChunker(used_file, max_llm_tokens=self.max_llm_tokens)
844
+ self.set_long_audio_protections(chunker.duration_ms)
845
+ logger.info(
846
+ "Long-audio transcription protections enabled: %s (duration: %sms, threshold: %sms)",
847
+ self.long_audio_protections_enabled,
848
+ chunker.duration_ms,
849
+ AUDIO_LONG_DURATION_THRESHOLD_MS,
850
+ )
652
851
  chunks = chunker.extract_chunks()
653
852
 
654
853
  logger.info(f"chunks: {chunks}")
@@ -9,8 +9,6 @@ from google.api_core import exceptions as google_exceptions
9
9
  from retry import retry
10
10
  from concurrent.futures import ThreadPoolExecutor, as_completed
11
11
 
12
- from polytext.processor.transcript_chunker import TranscriptChunker
13
- from polytext.processor.text_merger import TextMerger
14
12
  from polytext.prompts.beautiful_text import BEAUTIFUL_TEXT_PROMPT
15
13
 
16
14
  logger = logging.getLogger(__name__)
@@ -26,6 +24,7 @@ class BeautifulTextConverter:
26
24
  prompt_overhead: int = 1800,
27
25
  tokens_per_char: float = 0.25,
28
26
  overlap_chars: int = 800,
27
+ max_target_chars: int = 12000,
29
28
  ) -> None:
30
29
  self.llm_api_key = llm_api_key
31
30
  self.model = model
@@ -34,19 +33,58 @@ class BeautifulTextConverter:
34
33
  self.prompt_overhead = prompt_overhead
35
34
  self.tokens_per_char = tokens_per_char
36
35
  self.overlap_chars = overlap_chars
36
+ self.max_target_chars = max_target_chars
37
37
 
38
38
  def get_client(self):
39
39
  return genai.Client(api_key=self.llm_api_key) if self.llm_api_key else genai.Client()
40
40
 
41
41
  def chunk_raw_text(self, raw_text: str) -> list[dict]:
42
- chunker = TranscriptChunker(
43
- transcript=raw_text,
44
- max_llm_tokens=self.max_llm_tokens,
45
- prompt_overhead=self.prompt_overhead,
46
- tokens_per_char=self.tokens_per_char,
47
- overlap_chars=self.overlap_chars,
48
- )
49
- return chunker.chunk_transcript()
42
+ text = (raw_text or "").strip()
43
+ if not text:
44
+ return []
45
+
46
+ chunks = []
47
+ start = 0
48
+ index = 0
49
+
50
+ while start < len(text):
51
+ proposed_end = min(start + self.max_target_chars, len(text))
52
+ end = self._find_chunk_end(text, start, proposed_end)
53
+ target_text = text[start:end].strip()
54
+
55
+ if not target_text:
56
+ break
57
+
58
+ chunks.append(
59
+ {
60
+ "index": index,
61
+ "target_text": target_text,
62
+ }
63
+ )
64
+ start = end
65
+ index += 1
66
+
67
+ return chunks
68
+
69
+ def _find_chunk_end(self, text: str, start: int, proposed_end: int) -> int:
70
+ if proposed_end >= len(text):
71
+ return len(text)
72
+
73
+ minimum_end = start + int(self.max_target_chars * 0.65)
74
+ chunk_window = text[minimum_end:proposed_end]
75
+ sentence_boundaries = list(re.finditer(r"(?<=[.!?])\s+", chunk_window))
76
+ if sentence_boundaries:
77
+ return minimum_end + sentence_boundaries[-1].end()
78
+
79
+ paragraph_boundary = text.rfind("\n\n", minimum_end, proposed_end)
80
+ if paragraph_boundary != -1:
81
+ return paragraph_boundary + 2
82
+
83
+ whitespace_boundary = text.rfind(" ", minimum_end, proposed_end)
84
+ if whitespace_boundary != -1:
85
+ return whitespace_boundary + 1
86
+
87
+ return proposed_end
50
88
 
51
89
  @retry(
52
90
  (
@@ -60,11 +98,18 @@ class BeautifulTextConverter:
60
98
  backoff=2,
61
99
  logger=logger,
62
100
  )
63
- def process_chunk(self, client, chunk_text: str, index: int) -> dict:
101
+ def process_chunk(self, client, chunk: dict, index: int) -> dict:
64
102
  logger.info("Processing beautiful text chunk %s", index + 1)
65
103
  start_time = time.time()
66
104
 
105
+ target_text = chunk["target_text"]
106
+
67
107
  config = types.GenerateContentConfig(
108
+ temperature=0,
109
+ thinking_config=types.ThinkingConfig(thinking_budget=0),
110
+ max_output_tokens=self.max_llm_tokens,
111
+ tools=[],
112
+ automatic_function_calling=types.AutomaticFunctionCallingConfig(disable=True),
68
113
  safety_settings=[
69
114
  types.SafetySetting(
70
115
  category=types.HarmCategory.HARM_CATEGORY_HATE_SPEECH,
@@ -87,7 +132,11 @@ class BeautifulTextConverter:
87
132
 
88
133
  response = client.models.generate_content(
89
134
  model=self.model,
90
- contents=[BEAUTIFUL_TEXT_PROMPT, chunk_text],
135
+ contents=[
136
+ BEAUTIFUL_TEXT_PROMPT,
137
+ "TARGET TEXT TO CLEAN",
138
+ target_text,
139
+ ],
91
140
  config=config,
92
141
  )
93
142
 
@@ -100,7 +149,7 @@ class BeautifulTextConverter:
100
149
  }
101
150
 
102
151
  def merge_cleaned_chunks(self, chunks: list[str]) -> str:
103
- return TextMerger(llm_api_key=self.llm_api_key).merge_chunks(chunks=chunks)
152
+ return "\n\n".join(chunk.strip() for chunk in chunks if chunk.strip())
104
153
 
105
154
  def _convert_markdown_to_json(self, markdown_text: str) -> dict:
106
155
  if not markdown_text.strip():
@@ -181,7 +230,7 @@ class BeautifulTextConverter:
181
230
 
182
231
  with ThreadPoolExecutor() as executor:
183
232
  future_to_index = {
184
- executor.submit(self.process_chunk, client, chunk["text"], chunk["index"]): chunk["index"]
233
+ executor.submit(self.process_chunk, client, chunk, chunk["index"]): chunk["index"]
185
234
  for chunk in chunks
186
235
  }
187
236
 
@@ -242,7 +242,9 @@ class DocumentOCRToTextConverter:
242
242
  return error.code in OCR_RETRIABLE_OUTPUT_ERROR_CODES
243
243
 
244
244
  def should_fallback_temperature_retry(self, error: EmptyDocument, temperature: float) -> bool:
245
- expected_stage = 1 if self.markdown_output else 0
245
+ # The non-literal prompt retry advances every OCR mode to stage 1.
246
+ # Plain-text document OCR must therefore also try the fallback model at stage 1.
247
+ expected_stage = 1
246
248
  if self.fallback_stage != expected_stage:
247
249
  return False
248
250
  if error.code not in OCR_RETRIABLE_OUTPUT_ERROR_CODES:
@@ -368,6 +370,8 @@ class DocumentOCRToTextConverter:
368
370
  config = types.GenerateContentConfig(
369
371
  temperature=temperature,
370
372
  max_output_tokens=self.max_output_tokens,
373
+ tools=[],
374
+ automatic_function_calling=types.AutomaticFunctionCallingConfig(disable=True),
371
375
  safety_settings=[
372
376
  types.SafetySetting(
373
377
  category=types.HarmCategory.HARM_CATEGORY_HATE_SPEECH,
@@ -37,6 +37,14 @@ def has_consecutive_repetition(items: list[str], min_run_length: int = 3) -> boo
37
37
  return False
38
38
 
39
39
 
40
+ def has_excessive_consecutive_word_repetition(
41
+ text: str,
42
+ min_run_length: int = 12,
43
+ ) -> bool:
44
+ words = re.findall(r"\b\w+\b", (text or "").casefold())
45
+ return has_consecutive_repetition(words, min_run_length=min_run_length)
46
+
47
+
40
48
  def tail_has_excessive_repetition(
41
49
  text: str,
42
50
  tail_lines: int,
@@ -352,6 +352,8 @@ class OCRToTextConverter:
352
352
  config = types.GenerateContentConfig(
353
353
  temperature=temperature,
354
354
  max_output_tokens=self.max_output_tokens,
355
+ tools=[],
356
+ automatic_function_calling=types.AutomaticFunctionCallingConfig(disable=True),
355
357
  safety_settings=[
356
358
  types.SafetySetting(
357
359
  category=types.HarmCategory.HARM_CATEGORY_HATE_SPEECH,
@@ -141,6 +141,8 @@ class TextToMdConverter:
141
141
  start_time = time.time()
142
142
 
143
143
  config = types.GenerateContentConfig(
144
+ tools=[],
145
+ automatic_function_calling=types.AutomaticFunctionCallingConfig(disable=True),
144
146
  safety_settings=[
145
147
  types.SafetySetting(category=types.HarmCategory.HARM_CATEGORY_HATE_SPEECH, threshold=types.HarmBlockThreshold.BLOCK_NONE),
146
148
  types.SafetySetting(category=types.HarmCategory.HARM_CATEGORY_DANGEROUS_CONTENT, threshold=types.HarmBlockThreshold.BLOCK_NONE),
@@ -235,4 +237,4 @@ class TextToMdConverter:
235
237
 
236
238
  logging.info(f"**** FINISH ****")
237
239
 
238
- return result_dict
240
+ return result_dict