polytext 0.2.7__tar.gz → 0.2.8__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. {polytext-0.2.7 → polytext-0.2.8}/PKG-INFO +35 -1
  2. {polytext-0.2.7 → polytext-0.2.8}/README.md +34 -0
  3. {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/audio_to_text.py +116 -21
  4. {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/document_ocr_to_text.py +49 -9
  5. {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/ocr_to_text.py +50 -9
  6. {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/audio.py +7 -2
  7. polytext-0.2.8/polytext/loader/aws_auth.py +98 -0
  8. {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/base.py +20 -5
  9. {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/video.py +7 -2
  10. polytext-0.2.8/polytext/prompts/ocr.py +64 -0
  11. {polytext-0.2.7 → polytext-0.2.8}/polytext/prompts/transcription.py +217 -1
  12. {polytext-0.2.7 → polytext-0.2.8}/polytext.egg-info/PKG-INFO +35 -1
  13. {polytext-0.2.7 → polytext-0.2.8}/polytext.egg-info/SOURCES.txt +4 -0
  14. {polytext-0.2.7 → polytext-0.2.8}/setup.py +1 -1
  15. {polytext-0.2.7 → polytext-0.2.8}/tests/test_audio_transcription_model_migration.py +129 -16
  16. polytext-0.2.8/tests/test_aws_auth.py +193 -0
  17. {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_ocr_from_image.py +5 -5
  18. {polytext-0.2.7 → polytext-0.2.8}/tests/test_ocr_fallbacks.py +52 -13
  19. {polytext-0.2.7 → polytext-0.2.8}/tests/test_ocr_image_descriptions.py +11 -0
  20. polytext-0.2.8/tests/test_transcribe_s3_images_from_csv.py +234 -0
  21. polytext-0.2.8/tests/test_transcribe_s3_images_from_csv_script.py +278 -0
  22. polytext-0.2.7/polytext/prompts/ocr.py +0 -38
  23. {polytext-0.2.7 → polytext-0.2.8}/LICENSE +0 -0
  24. {polytext-0.2.7 → polytext-0.2.8}/polytext/__init__.py +0 -0
  25. {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/__init__.py +0 -0
  26. {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/base.py +0 -0
  27. {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/beautiful_text.py +0 -0
  28. {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/document_ocr_to_text_azure_oai.py +0 -0
  29. {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/gemini_quality_guards.py +0 -0
  30. {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/html_to_md.py +0 -0
  31. {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/md_to_text.py +0 -0
  32. {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/ocr_to_text_azure_oai.py +0 -0
  33. {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/pdf.py +0 -0
  34. {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/text_to_md.py +0 -0
  35. {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/video_to_audio.py +0 -0
  36. {polytext-0.2.7 → polytext-0.2.8}/polytext/exceptions/__init__.py +0 -0
  37. {polytext-0.2.7 → polytext-0.2.8}/polytext/exceptions/base.py +0 -0
  38. {polytext-0.2.7 → polytext-0.2.8}/polytext/generator/__init__.py +0 -0
  39. {polytext-0.2.7 → polytext-0.2.8}/polytext/generator/pdf.py +0 -0
  40. {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/__init__.py +0 -0
  41. {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/document.py +0 -0
  42. {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/document_ocr.py +0 -0
  43. {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/downloader/__init__.py +0 -0
  44. {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/downloader/downloader.py +0 -0
  45. {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/html.py +0 -0
  46. {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/markdown.py +0 -0
  47. {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/notebook.py +0 -0
  48. {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/ocr.py +0 -0
  49. {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/plain_text.py +0 -0
  50. {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/xml_xbrl.py +0 -0
  51. {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/youtube.py +0 -0
  52. {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/youtube_llm.py +0 -0
  53. {polytext-0.2.7 → polytext-0.2.8}/polytext/processor/__init__.py +0 -0
  54. {polytext-0.2.7 → polytext-0.2.8}/polytext/processor/audio_chunker.py +0 -0
  55. {polytext-0.2.7 → polytext-0.2.8}/polytext/processor/text_merger.py +0 -0
  56. {polytext-0.2.7 → polytext-0.2.8}/polytext/processor/transcript_chunker.py +0 -0
  57. {polytext-0.2.7 → polytext-0.2.8}/polytext/prompts/__init__.py +0 -0
  58. {polytext-0.2.7 → polytext-0.2.8}/polytext/prompts/beautiful_text.py +0 -0
  59. {polytext-0.2.7 → polytext-0.2.8}/polytext/prompts/text_merging.py +0 -0
  60. {polytext-0.2.7 → polytext-0.2.8}/polytext/prompts/text_to_md.py +0 -0
  61. {polytext-0.2.7 → polytext-0.2.8}/polytext/utils/__init__.py +0 -0
  62. {polytext-0.2.7 → polytext-0.2.8}/polytext/utils/utils.py +0 -0
  63. {polytext-0.2.7 → polytext-0.2.8}/polytext.egg-info/dependency_links.txt +0 -0
  64. {polytext-0.2.7 → polytext-0.2.8}/polytext.egg-info/not-zip-safe +0 -0
  65. {polytext-0.2.7 → polytext-0.2.8}/polytext.egg-info/requires.txt +0 -0
  66. {polytext-0.2.7 → polytext-0.2.8}/polytext.egg-info/top_level.txt +0 -0
  67. {polytext-0.2.7 → polytext-0.2.8}/pyproject.toml +0 -0
  68. {polytext-0.2.7 → polytext-0.2.8}/setup.cfg +0 -0
  69. {polytext-0.2.7 → polytext-0.2.8}/tests/test_audio_chunker.py +0 -0
  70. {polytext-0.2.7 → polytext-0.2.8}/tests/test_audio_comparison_helpers.py +0 -0
  71. {polytext-0.2.7 → polytext-0.2.8}/tests/test_base_loader_error_mapping.py +0 -0
  72. {polytext-0.2.7 → polytext-0.2.8}/tests/test_beautiful_text_manual.py +0 -0
  73. {polytext-0.2.7 → polytext-0.2.8}/tests/test_compare_audio_models.py +0 -0
  74. {polytext-0.2.7 → polytext-0.2.8}/tests/test_compare_document_ocr_to_text_models.py +0 -0
  75. {polytext-0.2.7 → polytext-0.2.8}/tests/test_compare_ocr_to_text_models.py +0 -0
  76. {polytext-0.2.7 → polytext-0.2.8}/tests/test_compare_youtube_models.py +0 -0
  77. {polytext-0.2.7 → polytext-0.2.8}/tests/test_dowload_audio_from_youtube.py +0 -0
  78. {polytext-0.2.7 → polytext-0.2.8}/tests/test_dowload_audio_from_youtube_helpers.py +0 -0
  79. {polytext-0.2.7 → polytext-0.2.8}/tests/test_extracted_text_whitespace.py +0 -0
  80. {polytext-0.2.7 → polytext-0.2.8}/tests/test_gemini_quality_guards.py +0 -0
  81. {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_audio_transcript_from_gcs.py +0 -0
  82. {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_customized_pdf_from_markdown.py +0 -0
  83. {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_document_ocr.py +0 -0
  84. {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_document_ocr_azure_oai.py +0 -0
  85. {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_document_text.py +0 -0
  86. {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_document_text_from_gcs.py +0 -0
  87. {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_text_from_markdown.py +0 -0
  88. {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_video_transcript_from_gcs.py +0 -0
  89. {polytext-0.2.7 → polytext-0.2.8}/tests/test_library.py +0 -0
  90. {polytext-0.2.7 → polytext-0.2.8}/tests/test_markdown_loader_gzip.py +0 -0
  91. {polytext-0.2.7 → polytext-0.2.8}/tests/test_markitdown_html.py +0 -0
  92. {polytext-0.2.7 → polytext-0.2.8}/tests/test_notebook_loader.py +0 -0
  93. {polytext-0.2.7 → polytext-0.2.8}/tests/test_pain_text.py +0 -0
  94. {polytext-0.2.7 → polytext-0.2.8}/tests/test_pdf_conversion_error.py +0 -0
  95. {polytext-0.2.7 → polytext-0.2.8}/tests/test_python_version_metadata.py +0 -0
  96. {polytext-0.2.7 → polytext-0.2.8}/tests/test_split_audio_with_llm.py +0 -0
  97. {polytext-0.2.7 → polytext-0.2.8}/tests/test_xml_xbrl_loader.py +0 -0
  98. {polytext-0.2.7 → polytext-0.2.8}/tests/test_youtube_gemini_minimal_check.py +0 -0
  99. {polytext-0.2.7 → polytext-0.2.8}/tests/test_youtube_llm_fallbacks.py +0 -0
  100. {polytext-0.2.7 → polytext-0.2.8}/tests/test_youtube_transcript.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: polytext
3
- Version: 0.2.7
3
+ Version: 0.2.8
4
4
  Summary: Python utilities to simplify document files management
5
5
  Home-page: https://github.com/docsity/polytext
6
6
  Author: Matteo Senardi
@@ -182,6 +182,40 @@ result = loader.get_text(input_list=["https://www.domain-name.com/path"])
182
182
  print(result["text"])
183
183
  ```
184
184
 
185
+ ### S3 authentication
186
+
187
+ By default, Polytext uses the standard boto3 credential chain when loading `s3://` inputs
188
+ (environment variables, AWS profiles, IAM roles, and other boto3-supported providers).
189
+
190
+ For runtimes that need to assume an AWS role through Google OIDC, STS web identity
191
+ authentication can be enabled explicitly:
192
+
193
+ ```python
194
+ from polytext.loader.base import BaseLoader
195
+
196
+ loader = BaseLoader(
197
+ aws_auth_mode="sts_web_identity",
198
+ aws_role_arn="arn:aws:iam::111122223333:role/ExampleRole",
199
+ aws_region="eu-central-1",
200
+ aws_role_session_name="polytext-session",
201
+ gcp_id_token_audience="example-gcp-audience",
202
+ )
203
+ ```
204
+
205
+ The same configuration can also come from environment variables:
206
+
207
+ ```bash
208
+ POLYTEXT_AWS_AUTH_MODE=sts_web_identity
209
+ AWS_ROLE_ARN=arn:aws:iam::111122223333:role/ExampleRole
210
+ AWS_REGION=eu-central-1
211
+ AWS_ROLE_SESSION_NAME=polytext-session
212
+ GCP_ID_TOKEN_AUDIENCE=example-gcp-audience
213
+ GOOGLE_APPLICATION_CREDENTIALS=/absolute/path/to/service_account.json
214
+ ```
215
+
216
+ Polytext uses the temporary STS credentials only to create the S3 client. It does not
217
+ export them to `os.environ` and does not reset boto3's global session.
218
+
185
219
  ## License
186
220
 
187
221
  MIT Licence
@@ -125,6 +125,40 @@ result = loader.get_text(input_list=["https://www.domain-name.com/path"])
125
125
  print(result["text"])
126
126
  ```
127
127
 
128
+ ### S3 authentication
129
+
130
+ By default, Polytext uses the standard boto3 credential chain when loading `s3://` inputs
131
+ (environment variables, AWS profiles, IAM roles, and other boto3-supported providers).
132
+
133
+ For runtimes that need to assume an AWS role through Google OIDC, STS web identity
134
+ authentication can be enabled explicitly:
135
+
136
+ ```python
137
+ from polytext.loader.base import BaseLoader
138
+
139
+ loader = BaseLoader(
140
+ aws_auth_mode="sts_web_identity",
141
+ aws_role_arn="arn:aws:iam::111122223333:role/ExampleRole",
142
+ aws_region="eu-central-1",
143
+ aws_role_session_name="polytext-session",
144
+ gcp_id_token_audience="example-gcp-audience",
145
+ )
146
+ ```
147
+
148
+ The same configuration can also come from environment variables:
149
+
150
+ ```bash
151
+ POLYTEXT_AWS_AUTH_MODE=sts_web_identity
152
+ AWS_ROLE_ARN=arn:aws:iam::111122223333:role/ExampleRole
153
+ AWS_REGION=eu-central-1
154
+ AWS_ROLE_SESSION_NAME=polytext-session
155
+ GCP_ID_TOKEN_AUDIENCE=example-gcp-audience
156
+ GOOGLE_APPLICATION_CREDENTIALS=/absolute/path/to/service_account.json
157
+ ```
158
+
159
+ Polytext uses the temporary STS credentials only to create the S3 client. It does not
160
+ export them to `os.environ` and does not reset boto3's global session.
161
+
128
162
  ## License
129
163
 
130
164
  MIT Licence
@@ -6,6 +6,7 @@ import time
6
6
  import mimetypes
7
7
  import uuid
8
8
  import re
9
+ import shutil
9
10
  import ffmpeg
10
11
  from retry import retry
11
12
  from google import genai
@@ -15,7 +16,13 @@ from concurrent.futures import ThreadPoolExecutor, as_completed
15
16
  from google.api_core import exceptions as google_exceptions
16
17
 
17
18
  from ..exceptions import EmptyDocument
18
- from ..prompts.transcription import AUDIO_TO_MARKDOWN_PROMPT, AUDIO_TO_PLAIN_TEXT_PROMPT
19
+ from ..prompts.transcription import (
20
+ AUDIO_TO_MARKDOWN_NON_LITERAL_FALLBACK_PROMPT,
21
+ AUDIO_TO_MARKDOWN_PROMPT,
22
+ AUDIO_TO_PLAIN_TEXT_PROMPT,
23
+ AUDIO_TO_MARKDOWN_RAW_NON_LITERAL_FALLBACK_PROMPT,
24
+ AUDIO_TO_MARKDOWN_PROMPT_IS_RAW,
25
+ )
19
26
  from ..processor.audio_chunker import AudioChunker
20
27
  from ..processor.text_merger import TextMerger
21
28
  from .gemini_quality_guards import extract_finish_reason, tail_has_excessive_repetition
@@ -48,6 +55,9 @@ AUDIO_FALLBACK_MODEL = os.getenv("AUDIO_FALLBACK_MODEL", "gemini-3-flash-preview
48
55
  AUDIO_FALLBACK_TEMPERATURE = float(os.getenv("AUDIO_FALLBACK_TEMPERATURE", "1.0"))
49
56
  AUDIO_FINAL_FALLBACK_MODEL = os.getenv("AUDIO_FINAL_FALLBACK_MODEL", "gemini-3.5-flash")
50
57
  AUDIO_FILE_UPLOAD_THRESHOLD_BYTES = 20 * 1024 * 1024
58
+ AUDIO_PROMPT_VARIANT_DEFAULT = "default"
59
+ AUDIO_PROMPT_VARIANT_NON_LITERAL_FALLBACK = "non_literal_fallback"
60
+ AUDIO_RETRIABLE_OUTPUT_ERROR_CODES = (996, 997, 999)
51
61
  NO_HUMAN_SPEECH_MARKER = "no human speech detected"
52
62
 
53
63
 
@@ -89,6 +99,21 @@ def add_line_break_after_each_sentence(text: str) -> str:
89
99
  return "\n".join(formatted_lines).strip()
90
100
 
91
101
 
102
+ def create_ascii_safe_upload_copy(audio_file: str) -> tuple[str, str | None]:
103
+ if os.path.basename(audio_file).isascii():
104
+ return audio_file, None
105
+
106
+ suffix = os.path.splitext(audio_file)[1]
107
+ if not suffix.isascii():
108
+ suffix = ""
109
+
110
+ fd, temp_upload_path = tempfile.mkstemp(prefix="audio-upload-", suffix=suffix)
111
+ os.close(fd)
112
+ if os.path.exists(audio_file):
113
+ shutil.copyfile(audio_file, temp_upload_path)
114
+ return temp_upload_path, temp_upload_path
115
+
116
+
92
117
  def compress_and_convert_audio(input_path: str, bitrate_quality: int = 9) -> str:
93
118
  """
94
119
  Compress and convert an audio file to MP3 using ffmpeg.
@@ -139,7 +164,8 @@ def transcribe_full_audio(audio_file, markdown_output: bool = False,
139
164
  save_transcript_chunks: bool = False, bitrate_quality=9,
140
165
  timeout_minutes: int = None,
141
166
  max_llm_tokens: int = 4250,
142
- max_output_tokens: int | None = None) -> dict:
167
+ max_output_tokens: int | None = None,
168
+ is_output_audio_raw: bool = True) -> dict:
143
169
  """
144
170
  Convenience function to transcribe an audio file into text, optionally formatted as Markdown.
145
171
 
@@ -157,13 +183,16 @@ def transcribe_full_audio(audio_file, markdown_output: bool = False,
157
183
  timeout_minutes (int, optional): Number of minutes to wait for a response. Defaults to None.
158
184
  max_llm_tokens (int, optional): Token budget used for audio chunk sizing. Defaults to 4250.
159
185
  max_output_tokens (int | None, optional): Maximum Gemini output tokens. Defaults to `max_llm_tokens`.
186
+ is_output_audio_raw (bool, optional): If True, use the raw Markdown audio prompt.
187
+ If False, use the formatted Markdown audio prompt. Defaults to True.
160
188
 
161
189
  Returns:
162
190
  str: The transcribed text from the audio file.
163
191
  """
164
192
  converter = AudioToTextConverter(markdown_output=markdown_output, llm_api_key=llm_api_key,
165
193
  bitrate_quality=bitrate_quality, timeout_minutes=timeout_minutes,
166
- max_llm_tokens=max_llm_tokens, max_output_tokens=max_output_tokens)
194
+ max_llm_tokens=max_llm_tokens, max_output_tokens=max_output_tokens,
195
+ is_output_audio_raw=is_output_audio_raw)
167
196
  return converter.transcribe_full_audio(audio_file, save_transcript_chunks)
168
197
 
169
198
 
@@ -173,7 +202,10 @@ class AudioToTextConverter:
173
202
  k: int = 5, min_matches: int = 3, markdown_output: bool = True, llm_api_key: str = None,
174
203
  max_llm_tokens: int = 4250,
175
204
  max_output_tokens: int | None = None, temp_dir: str = "temp",
176
- bitrate_quality: int = 9, timeout_minutes: int = None):
205
+ bitrate_quality: int = 9, timeout_minutes: int = None,
206
+ fallback_stage: int = 0,
207
+ prompt_variant: str = AUDIO_PROMPT_VARIANT_DEFAULT,
208
+ is_output_audio_raw: bool = True):
177
209
  """
178
210
  Initialize the AudioToTextConverter class with a specified transcription model and provider.
179
211
 
@@ -190,6 +222,12 @@ class AudioToTextConverter:
190
222
  temp_dir (str): Directory for temporary files. Defaults to "temp".
191
223
  bitrate_quality (int, optional): Variable bitrate quality from 0-9 (9 being lowest). Defaults to 9
192
224
  timeout_minutes (int): Number of minutes to wait for a response.
225
+ fallback_stage (int, optional): Internal retry stage used by fallback attempts.
226
+ Defaults to 0.
227
+ prompt_variant (str, optional): Prompt variant used by this attempt.
228
+ Defaults to "default".
229
+ is_output_audio_raw (bool, optional): If True, use the raw Markdown audio prompt.
230
+ If False, use the formatted Markdown audio prompt. Defaults to True.
193
231
 
194
232
  Raises:
195
233
  OSError: If temp directory creation fails
@@ -207,27 +245,56 @@ class AudioToTextConverter:
207
245
  self.chunked_audio = False
208
246
  self.bitrate_quality = bitrate_quality
209
247
  self.timeout_minutes = timeout_minutes
248
+ self.fallback_stage = fallback_stage
249
+ self.prompt_variant = prompt_variant
210
250
  self.fallback_source_pattern = AUDIO_FALLBACK_SOURCE_PATTERN
211
251
  self.fallback_model = AUDIO_FALLBACK_MODEL
212
252
  self.fallback_temperature = AUDIO_FALLBACK_TEMPERATURE
213
253
  self.final_fallback_model = AUDIO_FINAL_FALLBACK_MODEL
254
+ self.is_output_audio_raw = is_output_audio_raw
214
255
 
215
256
  # Set up custom temp directory
216
257
  self.temp_dir = os.path.abspath(temp_dir)
217
258
  os.makedirs(self.temp_dir, exist_ok=True)
218
259
  tempfile.tempdir = self.temp_dir
219
260
 
261
+ def _build_prompt_template(self) -> str:
262
+ if self.markdown_output and self.prompt_variant == AUDIO_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
263
+ if self.is_output_audio_raw:
264
+ return AUDIO_TO_MARKDOWN_RAW_NON_LITERAL_FALLBACK_PROMPT
265
+ return AUDIO_TO_MARKDOWN_NON_LITERAL_FALLBACK_PROMPT
266
+ if self.markdown_output and self.is_output_audio_raw:
267
+ return AUDIO_TO_MARKDOWN_PROMPT_IS_RAW
268
+ if self.markdown_output:
269
+ return AUDIO_TO_MARKDOWN_PROMPT
270
+ return AUDIO_TO_PLAIN_TEXT_PROMPT
271
+
272
+ def should_prompt_fallback_retry(self, error: EmptyDocument) -> bool:
273
+ if self.fallback_stage != 0:
274
+ return False
275
+ if not self.markdown_output:
276
+ return False
277
+ if self.prompt_variant == AUDIO_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
278
+ return False
279
+ return error.code in AUDIO_RETRIABLE_OUTPUT_ERROR_CODES
280
+
220
281
  def should_fallback_temperature_retry(self, error: EmptyDocument, temperature: float) -> bool:
221
- if error.code not in (996, 997, 999):
282
+ expected_stage = 1 if self.markdown_output else 0
283
+ if self.fallback_stage != expected_stage:
222
284
  return False
223
- if self.fallback_model == self.transcription_model and temperature == self.fallback_temperature:
285
+ if error.code not in AUDIO_RETRIABLE_OUTPUT_ERROR_CODES:
224
286
  return False
225
- if self.fallback_source_pattern not in self.transcription_model:
287
+ if self.fallback_model == self.transcription_model and temperature == self.fallback_temperature:
226
288
  return False
227
- return True
289
+ if self.fallback_source_pattern and self.fallback_source_pattern in self.transcription_model:
290
+ return True
291
+ return self.transcription_model != self.fallback_model
228
292
 
229
293
  def should_final_fallback_model(self, error: EmptyDocument) -> bool:
230
- if error.code not in (996, 997, 999):
294
+ expected_stage = 2 if self.markdown_output else 1
295
+ if self.fallback_stage != expected_stage:
296
+ return False
297
+ if error.code not in AUDIO_RETRIABLE_OUTPUT_ERROR_CODES:
231
298
  return False
232
299
  if self.final_fallback_model == self.transcription_model:
233
300
  return False
@@ -239,10 +306,14 @@ class AudioToTextConverter:
239
306
  reason: str,
240
307
  fallback_model: str,
241
308
  fallback_temperature: float,
309
+ fallback_stage: int,
310
+ prompt_variant: str | None = None,
242
311
  ) -> dict:
312
+ resolved_prompt_variant = prompt_variant or self.prompt_variant
243
313
  logger.info(
244
- "Retrying audio transcript with fallback model %s and temperature %s for %s because %s",
314
+ "Retrying audio transcript with fallback model %s, prompt variant %s and temperature %s for %s because %s",
245
315
  fallback_model,
316
+ resolved_prompt_variant,
246
317
  fallback_temperature,
247
318
  audio_file,
248
319
  reason,
@@ -259,15 +330,20 @@ class AudioToTextConverter:
259
330
  temp_dir=self.temp_dir,
260
331
  bitrate_quality=self.bitrate_quality,
261
332
  timeout_minutes=self.timeout_minutes,
333
+ fallback_stage=fallback_stage,
334
+ prompt_variant=resolved_prompt_variant,
335
+ is_output_audio_raw=self.is_output_audio_raw,
262
336
  )
263
337
  result = fallback_converter.transcribe_audio(
264
338
  audio_file=audio_file,
265
339
  temperature=fallback_temperature,
266
340
  )
267
- result["fallback_from_model"] = self.transcription_model
268
- result["fallback_to_model"] = fallback_model
269
- result["fallback_reason"] = reason
270
- result["fallback_temperature"] = fallback_temperature
341
+ result.setdefault("fallback_from_model", self.transcription_model)
342
+ result.setdefault("fallback_to_model", fallback_model)
343
+ result.setdefault("fallback_reason", reason)
344
+ result.setdefault("fallback_temperature", fallback_temperature)
345
+ result.setdefault("fallback_from_prompt_variant", self.prompt_variant)
346
+ result.setdefault("fallback_to_prompt_variant", resolved_prompt_variant)
271
347
  return result
272
348
 
273
349
  def build_config(self, output_budget: int, temperature: float = 0.0) -> types.GenerateContentConfig:
@@ -318,7 +394,8 @@ class AudioToTextConverter:
318
394
  if file_size > AUDIO_FILE_UPLOAD_THRESHOLD_BYTES:
319
395
  logger.info("Audio file size exceeds 20MB, uploading file before transcription")
320
396
 
321
- my_file = client.files.upload(file=audio_file)
397
+ upload_file, temp_upload_path = create_ascii_safe_upload_copy(audio_file)
398
+ my_file = client.files.upload(file=upload_file)
322
399
  try:
323
400
  response = client.models.count_tokens(
324
401
  model=self.transcription_model,
@@ -335,6 +412,8 @@ class AudioToTextConverter:
335
412
  )
336
413
  finally:
337
414
  client.files.delete(name=my_file.name)
415
+ if temp_upload_path and os.path.exists(temp_upload_path):
416
+ os.remove(temp_upload_path)
338
417
 
339
418
  logger.info("Audio file size does not exceed 20MB")
340
419
  with open(audio_file, "rb") as f:
@@ -397,14 +476,15 @@ class AudioToTextConverter:
397
476
 
398
477
  start_time = time.time()
399
478
 
479
+ prompt_template = self._build_prompt_template()
400
480
  if self.markdown_output:
401
- logger.info("Using prompt for markdown format")
402
- # Convert the text to Markdown format
403
- prompt_template = AUDIO_TO_MARKDOWN_PROMPT
481
+ logger.info(
482
+ "Using prompt for markdown format with variant %s and raw output %s",
483
+ self.prompt_variant,
484
+ self.is_output_audio_raw,
485
+ )
404
486
  else:
405
- logger.info("Using prompt for plain text format")
406
- # Convert the text to plain text format
407
- prompt_template = AUDIO_TO_PLAIN_TEXT_PROMPT
487
+ logger.info("Using prompt for plain text format with variant %s", self.prompt_variant)
408
488
 
409
489
  if self.llm_api_key:
410
490
  logger.info("Using provided Google API key")
@@ -469,6 +549,7 @@ class AudioToTextConverter:
469
549
  "finish_reason": finish_reason,
470
550
  "max_output_tokens": self.max_output_tokens,
471
551
  "temperature": temperature,
552
+ "prompt_variant": self.prompt_variant,
472
553
  }
473
554
 
474
555
  logger.info(
@@ -476,12 +557,22 @@ class AudioToTextConverter:
476
557
  )
477
558
  return response_dict
478
559
  except EmptyDocument as e:
560
+ if self.should_prompt_fallback_retry(e):
561
+ return self.run_fallback(
562
+ audio_file=audio_file,
563
+ reason=e.message,
564
+ fallback_model=self.transcription_model,
565
+ fallback_temperature=temperature,
566
+ fallback_stage=1,
567
+ prompt_variant=AUDIO_PROMPT_VARIANT_NON_LITERAL_FALLBACK,
568
+ )
479
569
  if self.should_fallback_temperature_retry(e, temperature):
480
570
  return self.run_fallback(
481
571
  audio_file=audio_file,
482
572
  reason=e.message,
483
573
  fallback_model=self.fallback_model,
484
574
  fallback_temperature=self.fallback_temperature,
575
+ fallback_stage=2 if self.markdown_output else 1,
485
576
  )
486
577
  if self.should_final_fallback_model(e):
487
578
  return self.run_fallback(
@@ -489,6 +580,7 @@ class AudioToTextConverter:
489
580
  reason=e.message,
490
581
  fallback_model=self.final_fallback_model,
491
582
  fallback_temperature=0.0,
583
+ fallback_stage=3 if self.markdown_output else 2,
492
584
  )
493
585
  raise
494
586
 
@@ -610,6 +702,9 @@ class AudioToTextConverter:
610
702
  "fallback_to_model",
611
703
  "fallback_reason",
612
704
  "fallback_temperature",
705
+ "prompt_variant",
706
+ "fallback_from_prompt_variant",
707
+ "fallback_to_prompt_variant",
613
708
  ):
614
709
  if key in chunk_results[0]:
615
710
  result_dict[key] = chunk_results[0][key]
@@ -11,6 +11,7 @@ from google.genai import types
11
11
  from google.api_core import exceptions as google_exceptions
12
12
 
13
13
  from ..prompts.ocr import (
14
+ OCR_TO_MARKDOWN_NON_LITERAL_FALLBACK_PROMPT,
14
15
  OCR_TO_MARKDOWN_PROMPT,
15
16
  OCR_TO_PLAIN_TEXT_PROMPT,
16
17
  build_ocr_prompt,
@@ -34,6 +35,9 @@ OCR_FALLBACK_SOURCE_PATTERN = os.getenv("OCR_FALLBACK_SOURCE_PATTERN", "flash-li
34
35
  OCR_FALLBACK_MODEL = os.getenv("OCR_FALLBACK_MODEL", "gemini-3-flash-preview")
35
36
  OCR_FALLBACK_TEMPERATURE = float(os.getenv("OCR_FALLBACK_TEMPERATURE", "1.0"))
36
37
  OCR_FINAL_FALLBACK_MODEL = os.getenv("OCR_FINAL_FALLBACK_MODEL", "gemini-2.0-flash")
38
+ OCR_PROMPT_VARIANT_DEFAULT = "default"
39
+ OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK = "non_literal_fallback"
40
+ OCR_RETRIABLE_OUTPUT_ERROR_CODES = (996, 997, 999)
37
41
 
38
42
 
39
43
  def compress_and_convert_image(input_path: str, target_size=1):
@@ -152,7 +156,8 @@ class DocumentOCRToTextConverter:
152
156
  def __init__(self, ocr_model="gemini-3.1-flash-lite", ocr_model_provider="google",
153
157
  markdown_output=True, llm_api_key=None, target_size=1, temp_dir="temp",
154
158
  page_range=None, timeout_minutes: int = None, fallback_stage: int = 0,
155
- max_output_tokens: int | None = None, include_image_descriptions: bool = False):
159
+ max_output_tokens: int | None = None, include_image_descriptions: bool = False,
160
+ prompt_variant: str = OCR_PROMPT_VARIANT_DEFAULT):
156
161
  """
157
162
  Initialize the DocumentOCRToTextConverter class with specified OCR model and formatting options.
158
163
 
@@ -175,6 +180,8 @@ class DocumentOCRToTextConverter:
175
180
  include_image_descriptions (bool, optional): If True, OCR prompts include
176
181
  brief functional descriptions for meaningful non-text images.
177
182
  Defaults to False.
183
+ prompt_variant (str, optional): Prompt variant used by this attempt.
184
+ Defaults to "default".
178
185
 
179
186
  Raises:
180
187
  OSError: If temp directory creation fails
@@ -188,6 +195,7 @@ class DocumentOCRToTextConverter:
188
195
  self.page_range = page_range
189
196
  self.timeout_minutes = timeout_minutes
190
197
  self.include_image_descriptions = include_image_descriptions
198
+ self.prompt_variant = prompt_variant
191
199
  requested_output_tokens = OCR_MAX_OUTPUT_TOKENS if max_output_tokens is None else max_output_tokens
192
200
  self.max_output_tokens = max(requested_output_tokens, OCR_MIN_OUTPUT_TOKENS)
193
201
  self.fallback_stage = fallback_stage
@@ -202,16 +210,31 @@ class DocumentOCRToTextConverter:
202
210
  tempfile.tempdir = self.temp_dir
203
211
 
204
212
  def _build_prompt_template(self) -> str:
205
- base_prompt = OCR_TO_MARKDOWN_PROMPT if self.markdown_output else OCR_TO_PLAIN_TEXT_PROMPT
213
+ if self.markdown_output and self.prompt_variant == OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
214
+ base_prompt = OCR_TO_MARKDOWN_NON_LITERAL_FALLBACK_PROMPT
215
+ elif self.markdown_output:
216
+ base_prompt = OCR_TO_MARKDOWN_PROMPT
217
+ else:
218
+ base_prompt = OCR_TO_PLAIN_TEXT_PROMPT
206
219
  return build_ocr_prompt(
207
220
  base_prompt,
208
221
  include_image_descriptions=self.include_image_descriptions,
209
222
  )
210
223
 
211
- def should_fallback_temperature_retry(self, error: EmptyDocument, temperature: float) -> bool:
224
+ def should_prompt_fallback_retry(self, error: EmptyDocument) -> bool:
212
225
  if self.fallback_stage != 0:
213
226
  return False
214
- if error.code not in (996, 997, 999):
227
+ if not self.markdown_output:
228
+ return False
229
+ if self.prompt_variant == OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
230
+ return False
231
+ return error.code in OCR_RETRIABLE_OUTPUT_ERROR_CODES
232
+
233
+ def should_fallback_temperature_retry(self, error: EmptyDocument, temperature: float) -> bool:
234
+ expected_stage = 1 if self.markdown_output else 0
235
+ if self.fallback_stage != expected_stage:
236
+ return False
237
+ if error.code not in OCR_RETRIABLE_OUTPUT_ERROR_CODES:
215
238
  return False
216
239
  if self.fallback_model == self.ocr_model and temperature == self.fallback_temperature:
217
240
  return False
@@ -220,9 +243,10 @@ class DocumentOCRToTextConverter:
220
243
  return self.ocr_model != self.fallback_model
221
244
 
222
245
  def should_final_fallback_model(self, error: EmptyDocument) -> bool:
223
- if self.fallback_stage != 1:
246
+ expected_stage = 2 if self.markdown_output else 1
247
+ if self.fallback_stage != expected_stage:
224
248
  return False
225
- if error.code not in (996, 997, 999):
249
+ if error.code not in OCR_RETRIABLE_OUTPUT_ERROR_CODES:
226
250
  return False
227
251
  if self.final_fallback_model == self.ocr_model:
228
252
  return False
@@ -235,10 +259,13 @@ class DocumentOCRToTextConverter:
235
259
  fallback_model: str,
236
260
  fallback_temperature: float,
237
261
  fallback_stage: int,
262
+ prompt_variant: str | None = None,
238
263
  ) -> dict:
264
+ resolved_prompt_variant = prompt_variant or self.prompt_variant
239
265
  logger.info(
240
- "Retrying document OCR with fallback model %s and temperature %s for %s because %s",
266
+ "Retrying document OCR with fallback model %s, prompt variant %s and temperature %s for %s because %s",
241
267
  fallback_model,
268
+ resolved_prompt_variant,
242
269
  fallback_temperature,
243
270
  file_for_ocr,
244
271
  reason,
@@ -255,6 +282,7 @@ class DocumentOCRToTextConverter:
255
282
  fallback_stage=fallback_stage,
256
283
  max_output_tokens=self.max_output_tokens,
257
284
  include_image_descriptions=self.include_image_descriptions,
285
+ prompt_variant=resolved_prompt_variant,
258
286
  )
259
287
  result = fallback_converter.get_ocr(
260
288
  file_for_ocr=file_for_ocr,
@@ -264,6 +292,8 @@ class DocumentOCRToTextConverter:
264
292
  result.setdefault("fallback_to_model", fallback_model)
265
293
  result.setdefault("fallback_reason", reason)
266
294
  result.setdefault("fallback_temperature", fallback_temperature)
295
+ result.setdefault("fallback_from_prompt_variant", self.prompt_variant)
296
+ result.setdefault("fallback_to_prompt_variant", resolved_prompt_variant)
267
297
  return result
268
298
 
269
299
  @retry(
@@ -448,18 +478,28 @@ class DocumentOCRToTextConverter:
448
478
  "finish_reason": finish_reason,
449
479
  "max_output_tokens": self.max_output_tokens,
450
480
  "temperature": temperature,
481
+ "prompt_variant": self.prompt_variant,
451
482
  }
452
483
 
453
484
  logger.info(f"OCR performed using {self.ocr_model} in {time_elapsed:.2f} seconds")
454
485
  return final_ocr_dict
455
486
  except EmptyDocument as e:
487
+ if self.should_prompt_fallback_retry(e):
488
+ return self.run_fallback(
489
+ file_for_ocr=file_for_ocr,
490
+ reason=e.message,
491
+ fallback_model=self.ocr_model,
492
+ fallback_temperature=temperature,
493
+ fallback_stage=1,
494
+ prompt_variant=OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK,
495
+ )
456
496
  if self.should_fallback_temperature_retry(e, temperature):
457
497
  return self.run_fallback(
458
498
  file_for_ocr=file_for_ocr,
459
499
  reason=e.message,
460
500
  fallback_model=self.fallback_model,
461
501
  fallback_temperature=self.fallback_temperature,
462
- fallback_stage=1,
502
+ fallback_stage=2 if self.markdown_output else 1,
463
503
  )
464
504
  if self.should_final_fallback_model(e):
465
505
  return self.run_fallback(
@@ -467,7 +507,7 @@ class DocumentOCRToTextConverter:
467
507
  reason=e.message,
468
508
  fallback_model=self.final_fallback_model,
469
509
  fallback_temperature=0.0,
470
- fallback_stage=2,
510
+ fallback_stage=3 if self.markdown_output else 2,
471
511
  )
472
512
  raise
473
513