polytext 0.2.7__tar.gz → 0.2.8__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {polytext-0.2.7 → polytext-0.2.8}/PKG-INFO +35 -1
- {polytext-0.2.7 → polytext-0.2.8}/README.md +34 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/audio_to_text.py +116 -21
- {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/document_ocr_to_text.py +49 -9
- {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/ocr_to_text.py +50 -9
- {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/audio.py +7 -2
- polytext-0.2.8/polytext/loader/aws_auth.py +98 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/base.py +20 -5
- {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/video.py +7 -2
- polytext-0.2.8/polytext/prompts/ocr.py +64 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/prompts/transcription.py +217 -1
- {polytext-0.2.7 → polytext-0.2.8}/polytext.egg-info/PKG-INFO +35 -1
- {polytext-0.2.7 → polytext-0.2.8}/polytext.egg-info/SOURCES.txt +4 -0
- {polytext-0.2.7 → polytext-0.2.8}/setup.py +1 -1
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_audio_transcription_model_migration.py +129 -16
- polytext-0.2.8/tests/test_aws_auth.py +193 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_ocr_from_image.py +5 -5
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_ocr_fallbacks.py +52 -13
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_ocr_image_descriptions.py +11 -0
- polytext-0.2.8/tests/test_transcribe_s3_images_from_csv.py +234 -0
- polytext-0.2.8/tests/test_transcribe_s3_images_from_csv_script.py +278 -0
- polytext-0.2.7/polytext/prompts/ocr.py +0 -38
- {polytext-0.2.7 → polytext-0.2.8}/LICENSE +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/__init__.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/__init__.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/base.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/beautiful_text.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/document_ocr_to_text_azure_oai.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/gemini_quality_guards.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/html_to_md.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/md_to_text.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/ocr_to_text_azure_oai.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/pdf.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/text_to_md.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/converter/video_to_audio.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/exceptions/__init__.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/exceptions/base.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/generator/__init__.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/generator/pdf.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/__init__.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/document.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/document_ocr.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/downloader/__init__.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/downloader/downloader.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/html.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/markdown.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/notebook.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/ocr.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/plain_text.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/xml_xbrl.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/youtube.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/loader/youtube_llm.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/processor/__init__.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/processor/audio_chunker.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/processor/text_merger.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/processor/transcript_chunker.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/prompts/__init__.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/prompts/beautiful_text.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/prompts/text_merging.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/prompts/text_to_md.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/utils/__init__.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext/utils/utils.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext.egg-info/dependency_links.txt +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext.egg-info/not-zip-safe +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext.egg-info/requires.txt +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/polytext.egg-info/top_level.txt +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/pyproject.toml +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/setup.cfg +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_audio_chunker.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_audio_comparison_helpers.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_base_loader_error_mapping.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_beautiful_text_manual.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_compare_audio_models.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_compare_document_ocr_to_text_models.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_compare_ocr_to_text_models.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_compare_youtube_models.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_dowload_audio_from_youtube.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_dowload_audio_from_youtube_helpers.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_extracted_text_whitespace.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_gemini_quality_guards.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_audio_transcript_from_gcs.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_customized_pdf_from_markdown.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_document_ocr.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_document_ocr_azure_oai.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_document_text.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_document_text_from_gcs.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_text_from_markdown.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_get_video_transcript_from_gcs.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_library.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_markdown_loader_gzip.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_markitdown_html.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_notebook_loader.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_pain_text.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_pdf_conversion_error.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_python_version_metadata.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_split_audio_with_llm.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_xml_xbrl_loader.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_youtube_gemini_minimal_check.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_youtube_llm_fallbacks.py +0 -0
- {polytext-0.2.7 → polytext-0.2.8}/tests/test_youtube_transcript.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: polytext
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.8
|
|
4
4
|
Summary: Python utilities to simplify document files management
|
|
5
5
|
Home-page: https://github.com/docsity/polytext
|
|
6
6
|
Author: Matteo Senardi
|
|
@@ -182,6 +182,40 @@ result = loader.get_text(input_list=["https://www.domain-name.com/path"])
|
|
|
182
182
|
print(result["text"])
|
|
183
183
|
```
|
|
184
184
|
|
|
185
|
+
### S3 authentication
|
|
186
|
+
|
|
187
|
+
By default, Polytext uses the standard boto3 credential chain when loading `s3://` inputs
|
|
188
|
+
(environment variables, AWS profiles, IAM roles, and other boto3-supported providers).
|
|
189
|
+
|
|
190
|
+
For runtimes that need to assume an AWS role through Google OIDC, STS web identity
|
|
191
|
+
authentication can be enabled explicitly:
|
|
192
|
+
|
|
193
|
+
```python
|
|
194
|
+
from polytext.loader.base import BaseLoader
|
|
195
|
+
|
|
196
|
+
loader = BaseLoader(
|
|
197
|
+
aws_auth_mode="sts_web_identity",
|
|
198
|
+
aws_role_arn="arn:aws:iam::111122223333:role/ExampleRole",
|
|
199
|
+
aws_region="eu-central-1",
|
|
200
|
+
aws_role_session_name="polytext-session",
|
|
201
|
+
gcp_id_token_audience="example-gcp-audience",
|
|
202
|
+
)
|
|
203
|
+
```
|
|
204
|
+
|
|
205
|
+
The same configuration can also come from environment variables:
|
|
206
|
+
|
|
207
|
+
```bash
|
|
208
|
+
POLYTEXT_AWS_AUTH_MODE=sts_web_identity
|
|
209
|
+
AWS_ROLE_ARN=arn:aws:iam::111122223333:role/ExampleRole
|
|
210
|
+
AWS_REGION=eu-central-1
|
|
211
|
+
AWS_ROLE_SESSION_NAME=polytext-session
|
|
212
|
+
GCP_ID_TOKEN_AUDIENCE=example-gcp-audience
|
|
213
|
+
GOOGLE_APPLICATION_CREDENTIALS=/absolute/path/to/service_account.json
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
Polytext uses the temporary STS credentials only to create the S3 client. It does not
|
|
217
|
+
export them to `os.environ` and does not reset boto3's global session.
|
|
218
|
+
|
|
185
219
|
## License
|
|
186
220
|
|
|
187
221
|
MIT Licence
|
|
@@ -125,6 +125,40 @@ result = loader.get_text(input_list=["https://www.domain-name.com/path"])
|
|
|
125
125
|
print(result["text"])
|
|
126
126
|
```
|
|
127
127
|
|
|
128
|
+
### S3 authentication
|
|
129
|
+
|
|
130
|
+
By default, Polytext uses the standard boto3 credential chain when loading `s3://` inputs
|
|
131
|
+
(environment variables, AWS profiles, IAM roles, and other boto3-supported providers).
|
|
132
|
+
|
|
133
|
+
For runtimes that need to assume an AWS role through Google OIDC, STS web identity
|
|
134
|
+
authentication can be enabled explicitly:
|
|
135
|
+
|
|
136
|
+
```python
|
|
137
|
+
from polytext.loader.base import BaseLoader
|
|
138
|
+
|
|
139
|
+
loader = BaseLoader(
|
|
140
|
+
aws_auth_mode="sts_web_identity",
|
|
141
|
+
aws_role_arn="arn:aws:iam::111122223333:role/ExampleRole",
|
|
142
|
+
aws_region="eu-central-1",
|
|
143
|
+
aws_role_session_name="polytext-session",
|
|
144
|
+
gcp_id_token_audience="example-gcp-audience",
|
|
145
|
+
)
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
The same configuration can also come from environment variables:
|
|
149
|
+
|
|
150
|
+
```bash
|
|
151
|
+
POLYTEXT_AWS_AUTH_MODE=sts_web_identity
|
|
152
|
+
AWS_ROLE_ARN=arn:aws:iam::111122223333:role/ExampleRole
|
|
153
|
+
AWS_REGION=eu-central-1
|
|
154
|
+
AWS_ROLE_SESSION_NAME=polytext-session
|
|
155
|
+
GCP_ID_TOKEN_AUDIENCE=example-gcp-audience
|
|
156
|
+
GOOGLE_APPLICATION_CREDENTIALS=/absolute/path/to/service_account.json
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
Polytext uses the temporary STS credentials only to create the S3 client. It does not
|
|
160
|
+
export them to `os.environ` and does not reset boto3's global session.
|
|
161
|
+
|
|
128
162
|
## License
|
|
129
163
|
|
|
130
164
|
MIT Licence
|
|
@@ -6,6 +6,7 @@ import time
|
|
|
6
6
|
import mimetypes
|
|
7
7
|
import uuid
|
|
8
8
|
import re
|
|
9
|
+
import shutil
|
|
9
10
|
import ffmpeg
|
|
10
11
|
from retry import retry
|
|
11
12
|
from google import genai
|
|
@@ -15,7 +16,13 @@ from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
|
15
16
|
from google.api_core import exceptions as google_exceptions
|
|
16
17
|
|
|
17
18
|
from ..exceptions import EmptyDocument
|
|
18
|
-
from ..prompts.transcription import
|
|
19
|
+
from ..prompts.transcription import (
|
|
20
|
+
AUDIO_TO_MARKDOWN_NON_LITERAL_FALLBACK_PROMPT,
|
|
21
|
+
AUDIO_TO_MARKDOWN_PROMPT,
|
|
22
|
+
AUDIO_TO_PLAIN_TEXT_PROMPT,
|
|
23
|
+
AUDIO_TO_MARKDOWN_RAW_NON_LITERAL_FALLBACK_PROMPT,
|
|
24
|
+
AUDIO_TO_MARKDOWN_PROMPT_IS_RAW,
|
|
25
|
+
)
|
|
19
26
|
from ..processor.audio_chunker import AudioChunker
|
|
20
27
|
from ..processor.text_merger import TextMerger
|
|
21
28
|
from .gemini_quality_guards import extract_finish_reason, tail_has_excessive_repetition
|
|
@@ -48,6 +55,9 @@ AUDIO_FALLBACK_MODEL = os.getenv("AUDIO_FALLBACK_MODEL", "gemini-3-flash-preview
|
|
|
48
55
|
AUDIO_FALLBACK_TEMPERATURE = float(os.getenv("AUDIO_FALLBACK_TEMPERATURE", "1.0"))
|
|
49
56
|
AUDIO_FINAL_FALLBACK_MODEL = os.getenv("AUDIO_FINAL_FALLBACK_MODEL", "gemini-3.5-flash")
|
|
50
57
|
AUDIO_FILE_UPLOAD_THRESHOLD_BYTES = 20 * 1024 * 1024
|
|
58
|
+
AUDIO_PROMPT_VARIANT_DEFAULT = "default"
|
|
59
|
+
AUDIO_PROMPT_VARIANT_NON_LITERAL_FALLBACK = "non_literal_fallback"
|
|
60
|
+
AUDIO_RETRIABLE_OUTPUT_ERROR_CODES = (996, 997, 999)
|
|
51
61
|
NO_HUMAN_SPEECH_MARKER = "no human speech detected"
|
|
52
62
|
|
|
53
63
|
|
|
@@ -89,6 +99,21 @@ def add_line_break_after_each_sentence(text: str) -> str:
|
|
|
89
99
|
return "\n".join(formatted_lines).strip()
|
|
90
100
|
|
|
91
101
|
|
|
102
|
+
def create_ascii_safe_upload_copy(audio_file: str) -> tuple[str, str | None]:
|
|
103
|
+
if os.path.basename(audio_file).isascii():
|
|
104
|
+
return audio_file, None
|
|
105
|
+
|
|
106
|
+
suffix = os.path.splitext(audio_file)[1]
|
|
107
|
+
if not suffix.isascii():
|
|
108
|
+
suffix = ""
|
|
109
|
+
|
|
110
|
+
fd, temp_upload_path = tempfile.mkstemp(prefix="audio-upload-", suffix=suffix)
|
|
111
|
+
os.close(fd)
|
|
112
|
+
if os.path.exists(audio_file):
|
|
113
|
+
shutil.copyfile(audio_file, temp_upload_path)
|
|
114
|
+
return temp_upload_path, temp_upload_path
|
|
115
|
+
|
|
116
|
+
|
|
92
117
|
def compress_and_convert_audio(input_path: str, bitrate_quality: int = 9) -> str:
|
|
93
118
|
"""
|
|
94
119
|
Compress and convert an audio file to MP3 using ffmpeg.
|
|
@@ -139,7 +164,8 @@ def transcribe_full_audio(audio_file, markdown_output: bool = False,
|
|
|
139
164
|
save_transcript_chunks: bool = False, bitrate_quality=9,
|
|
140
165
|
timeout_minutes: int = None,
|
|
141
166
|
max_llm_tokens: int = 4250,
|
|
142
|
-
max_output_tokens: int | None = None
|
|
167
|
+
max_output_tokens: int | None = None,
|
|
168
|
+
is_output_audio_raw: bool = True) -> dict:
|
|
143
169
|
"""
|
|
144
170
|
Convenience function to transcribe an audio file into text, optionally formatted as Markdown.
|
|
145
171
|
|
|
@@ -157,13 +183,16 @@ def transcribe_full_audio(audio_file, markdown_output: bool = False,
|
|
|
157
183
|
timeout_minutes (int, optional): Number of minutes to wait for a response. Defaults to None.
|
|
158
184
|
max_llm_tokens (int, optional): Token budget used for audio chunk sizing. Defaults to 4250.
|
|
159
185
|
max_output_tokens (int | None, optional): Maximum Gemini output tokens. Defaults to `max_llm_tokens`.
|
|
186
|
+
is_output_audio_raw (bool, optional): If True, use the raw Markdown audio prompt.
|
|
187
|
+
If False, use the formatted Markdown audio prompt. Defaults to True.
|
|
160
188
|
|
|
161
189
|
Returns:
|
|
162
190
|
str: The transcribed text from the audio file.
|
|
163
191
|
"""
|
|
164
192
|
converter = AudioToTextConverter(markdown_output=markdown_output, llm_api_key=llm_api_key,
|
|
165
193
|
bitrate_quality=bitrate_quality, timeout_minutes=timeout_minutes,
|
|
166
|
-
max_llm_tokens=max_llm_tokens, max_output_tokens=max_output_tokens
|
|
194
|
+
max_llm_tokens=max_llm_tokens, max_output_tokens=max_output_tokens,
|
|
195
|
+
is_output_audio_raw=is_output_audio_raw)
|
|
167
196
|
return converter.transcribe_full_audio(audio_file, save_transcript_chunks)
|
|
168
197
|
|
|
169
198
|
|
|
@@ -173,7 +202,10 @@ class AudioToTextConverter:
|
|
|
173
202
|
k: int = 5, min_matches: int = 3, markdown_output: bool = True, llm_api_key: str = None,
|
|
174
203
|
max_llm_tokens: int = 4250,
|
|
175
204
|
max_output_tokens: int | None = None, temp_dir: str = "temp",
|
|
176
|
-
bitrate_quality: int = 9, timeout_minutes: int = None
|
|
205
|
+
bitrate_quality: int = 9, timeout_minutes: int = None,
|
|
206
|
+
fallback_stage: int = 0,
|
|
207
|
+
prompt_variant: str = AUDIO_PROMPT_VARIANT_DEFAULT,
|
|
208
|
+
is_output_audio_raw: bool = True):
|
|
177
209
|
"""
|
|
178
210
|
Initialize the AudioToTextConverter class with a specified transcription model and provider.
|
|
179
211
|
|
|
@@ -190,6 +222,12 @@ class AudioToTextConverter:
|
|
|
190
222
|
temp_dir (str): Directory for temporary files. Defaults to "temp".
|
|
191
223
|
bitrate_quality (int, optional): Variable bitrate quality from 0-9 (9 being lowest). Defaults to 9
|
|
192
224
|
timeout_minutes (int): Number of minutes to wait for a response.
|
|
225
|
+
fallback_stage (int, optional): Internal retry stage used by fallback attempts.
|
|
226
|
+
Defaults to 0.
|
|
227
|
+
prompt_variant (str, optional): Prompt variant used by this attempt.
|
|
228
|
+
Defaults to "default".
|
|
229
|
+
is_output_audio_raw (bool, optional): If True, use the raw Markdown audio prompt.
|
|
230
|
+
If False, use the formatted Markdown audio prompt. Defaults to True.
|
|
193
231
|
|
|
194
232
|
Raises:
|
|
195
233
|
OSError: If temp directory creation fails
|
|
@@ -207,27 +245,56 @@ class AudioToTextConverter:
|
|
|
207
245
|
self.chunked_audio = False
|
|
208
246
|
self.bitrate_quality = bitrate_quality
|
|
209
247
|
self.timeout_minutes = timeout_minutes
|
|
248
|
+
self.fallback_stage = fallback_stage
|
|
249
|
+
self.prompt_variant = prompt_variant
|
|
210
250
|
self.fallback_source_pattern = AUDIO_FALLBACK_SOURCE_PATTERN
|
|
211
251
|
self.fallback_model = AUDIO_FALLBACK_MODEL
|
|
212
252
|
self.fallback_temperature = AUDIO_FALLBACK_TEMPERATURE
|
|
213
253
|
self.final_fallback_model = AUDIO_FINAL_FALLBACK_MODEL
|
|
254
|
+
self.is_output_audio_raw = is_output_audio_raw
|
|
214
255
|
|
|
215
256
|
# Set up custom temp directory
|
|
216
257
|
self.temp_dir = os.path.abspath(temp_dir)
|
|
217
258
|
os.makedirs(self.temp_dir, exist_ok=True)
|
|
218
259
|
tempfile.tempdir = self.temp_dir
|
|
219
260
|
|
|
261
|
+
def _build_prompt_template(self) -> str:
|
|
262
|
+
if self.markdown_output and self.prompt_variant == AUDIO_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
|
|
263
|
+
if self.is_output_audio_raw:
|
|
264
|
+
return AUDIO_TO_MARKDOWN_RAW_NON_LITERAL_FALLBACK_PROMPT
|
|
265
|
+
return AUDIO_TO_MARKDOWN_NON_LITERAL_FALLBACK_PROMPT
|
|
266
|
+
if self.markdown_output and self.is_output_audio_raw:
|
|
267
|
+
return AUDIO_TO_MARKDOWN_PROMPT_IS_RAW
|
|
268
|
+
if self.markdown_output:
|
|
269
|
+
return AUDIO_TO_MARKDOWN_PROMPT
|
|
270
|
+
return AUDIO_TO_PLAIN_TEXT_PROMPT
|
|
271
|
+
|
|
272
|
+
def should_prompt_fallback_retry(self, error: EmptyDocument) -> bool:
|
|
273
|
+
if self.fallback_stage != 0:
|
|
274
|
+
return False
|
|
275
|
+
if not self.markdown_output:
|
|
276
|
+
return False
|
|
277
|
+
if self.prompt_variant == AUDIO_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
|
|
278
|
+
return False
|
|
279
|
+
return error.code in AUDIO_RETRIABLE_OUTPUT_ERROR_CODES
|
|
280
|
+
|
|
220
281
|
def should_fallback_temperature_retry(self, error: EmptyDocument, temperature: float) -> bool:
|
|
221
|
-
|
|
282
|
+
expected_stage = 1 if self.markdown_output else 0
|
|
283
|
+
if self.fallback_stage != expected_stage:
|
|
222
284
|
return False
|
|
223
|
-
if
|
|
285
|
+
if error.code not in AUDIO_RETRIABLE_OUTPUT_ERROR_CODES:
|
|
224
286
|
return False
|
|
225
|
-
if self.
|
|
287
|
+
if self.fallback_model == self.transcription_model and temperature == self.fallback_temperature:
|
|
226
288
|
return False
|
|
227
|
-
|
|
289
|
+
if self.fallback_source_pattern and self.fallback_source_pattern in self.transcription_model:
|
|
290
|
+
return True
|
|
291
|
+
return self.transcription_model != self.fallback_model
|
|
228
292
|
|
|
229
293
|
def should_final_fallback_model(self, error: EmptyDocument) -> bool:
|
|
230
|
-
|
|
294
|
+
expected_stage = 2 if self.markdown_output else 1
|
|
295
|
+
if self.fallback_stage != expected_stage:
|
|
296
|
+
return False
|
|
297
|
+
if error.code not in AUDIO_RETRIABLE_OUTPUT_ERROR_CODES:
|
|
231
298
|
return False
|
|
232
299
|
if self.final_fallback_model == self.transcription_model:
|
|
233
300
|
return False
|
|
@@ -239,10 +306,14 @@ class AudioToTextConverter:
|
|
|
239
306
|
reason: str,
|
|
240
307
|
fallback_model: str,
|
|
241
308
|
fallback_temperature: float,
|
|
309
|
+
fallback_stage: int,
|
|
310
|
+
prompt_variant: str | None = None,
|
|
242
311
|
) -> dict:
|
|
312
|
+
resolved_prompt_variant = prompt_variant or self.prompt_variant
|
|
243
313
|
logger.info(
|
|
244
|
-
"Retrying audio transcript with fallback model %s and temperature %s for %s because %s",
|
|
314
|
+
"Retrying audio transcript with fallback model %s, prompt variant %s and temperature %s for %s because %s",
|
|
245
315
|
fallback_model,
|
|
316
|
+
resolved_prompt_variant,
|
|
246
317
|
fallback_temperature,
|
|
247
318
|
audio_file,
|
|
248
319
|
reason,
|
|
@@ -259,15 +330,20 @@ class AudioToTextConverter:
|
|
|
259
330
|
temp_dir=self.temp_dir,
|
|
260
331
|
bitrate_quality=self.bitrate_quality,
|
|
261
332
|
timeout_minutes=self.timeout_minutes,
|
|
333
|
+
fallback_stage=fallback_stage,
|
|
334
|
+
prompt_variant=resolved_prompt_variant,
|
|
335
|
+
is_output_audio_raw=self.is_output_audio_raw,
|
|
262
336
|
)
|
|
263
337
|
result = fallback_converter.transcribe_audio(
|
|
264
338
|
audio_file=audio_file,
|
|
265
339
|
temperature=fallback_temperature,
|
|
266
340
|
)
|
|
267
|
-
result
|
|
268
|
-
result
|
|
269
|
-
result
|
|
270
|
-
result
|
|
341
|
+
result.setdefault("fallback_from_model", self.transcription_model)
|
|
342
|
+
result.setdefault("fallback_to_model", fallback_model)
|
|
343
|
+
result.setdefault("fallback_reason", reason)
|
|
344
|
+
result.setdefault("fallback_temperature", fallback_temperature)
|
|
345
|
+
result.setdefault("fallback_from_prompt_variant", self.prompt_variant)
|
|
346
|
+
result.setdefault("fallback_to_prompt_variant", resolved_prompt_variant)
|
|
271
347
|
return result
|
|
272
348
|
|
|
273
349
|
def build_config(self, output_budget: int, temperature: float = 0.0) -> types.GenerateContentConfig:
|
|
@@ -318,7 +394,8 @@ class AudioToTextConverter:
|
|
|
318
394
|
if file_size > AUDIO_FILE_UPLOAD_THRESHOLD_BYTES:
|
|
319
395
|
logger.info("Audio file size exceeds 20MB, uploading file before transcription")
|
|
320
396
|
|
|
321
|
-
|
|
397
|
+
upload_file, temp_upload_path = create_ascii_safe_upload_copy(audio_file)
|
|
398
|
+
my_file = client.files.upload(file=upload_file)
|
|
322
399
|
try:
|
|
323
400
|
response = client.models.count_tokens(
|
|
324
401
|
model=self.transcription_model,
|
|
@@ -335,6 +412,8 @@ class AudioToTextConverter:
|
|
|
335
412
|
)
|
|
336
413
|
finally:
|
|
337
414
|
client.files.delete(name=my_file.name)
|
|
415
|
+
if temp_upload_path and os.path.exists(temp_upload_path):
|
|
416
|
+
os.remove(temp_upload_path)
|
|
338
417
|
|
|
339
418
|
logger.info("Audio file size does not exceed 20MB")
|
|
340
419
|
with open(audio_file, "rb") as f:
|
|
@@ -397,14 +476,15 @@ class AudioToTextConverter:
|
|
|
397
476
|
|
|
398
477
|
start_time = time.time()
|
|
399
478
|
|
|
479
|
+
prompt_template = self._build_prompt_template()
|
|
400
480
|
if self.markdown_output:
|
|
401
|
-
logger.info(
|
|
402
|
-
|
|
403
|
-
|
|
481
|
+
logger.info(
|
|
482
|
+
"Using prompt for markdown format with variant %s and raw output %s",
|
|
483
|
+
self.prompt_variant,
|
|
484
|
+
self.is_output_audio_raw,
|
|
485
|
+
)
|
|
404
486
|
else:
|
|
405
|
-
logger.info("Using prompt for plain text format")
|
|
406
|
-
# Convert the text to plain text format
|
|
407
|
-
prompt_template = AUDIO_TO_PLAIN_TEXT_PROMPT
|
|
487
|
+
logger.info("Using prompt for plain text format with variant %s", self.prompt_variant)
|
|
408
488
|
|
|
409
489
|
if self.llm_api_key:
|
|
410
490
|
logger.info("Using provided Google API key")
|
|
@@ -469,6 +549,7 @@ class AudioToTextConverter:
|
|
|
469
549
|
"finish_reason": finish_reason,
|
|
470
550
|
"max_output_tokens": self.max_output_tokens,
|
|
471
551
|
"temperature": temperature,
|
|
552
|
+
"prompt_variant": self.prompt_variant,
|
|
472
553
|
}
|
|
473
554
|
|
|
474
555
|
logger.info(
|
|
@@ -476,12 +557,22 @@ class AudioToTextConverter:
|
|
|
476
557
|
)
|
|
477
558
|
return response_dict
|
|
478
559
|
except EmptyDocument as e:
|
|
560
|
+
if self.should_prompt_fallback_retry(e):
|
|
561
|
+
return self.run_fallback(
|
|
562
|
+
audio_file=audio_file,
|
|
563
|
+
reason=e.message,
|
|
564
|
+
fallback_model=self.transcription_model,
|
|
565
|
+
fallback_temperature=temperature,
|
|
566
|
+
fallback_stage=1,
|
|
567
|
+
prompt_variant=AUDIO_PROMPT_VARIANT_NON_LITERAL_FALLBACK,
|
|
568
|
+
)
|
|
479
569
|
if self.should_fallback_temperature_retry(e, temperature):
|
|
480
570
|
return self.run_fallback(
|
|
481
571
|
audio_file=audio_file,
|
|
482
572
|
reason=e.message,
|
|
483
573
|
fallback_model=self.fallback_model,
|
|
484
574
|
fallback_temperature=self.fallback_temperature,
|
|
575
|
+
fallback_stage=2 if self.markdown_output else 1,
|
|
485
576
|
)
|
|
486
577
|
if self.should_final_fallback_model(e):
|
|
487
578
|
return self.run_fallback(
|
|
@@ -489,6 +580,7 @@ class AudioToTextConverter:
|
|
|
489
580
|
reason=e.message,
|
|
490
581
|
fallback_model=self.final_fallback_model,
|
|
491
582
|
fallback_temperature=0.0,
|
|
583
|
+
fallback_stage=3 if self.markdown_output else 2,
|
|
492
584
|
)
|
|
493
585
|
raise
|
|
494
586
|
|
|
@@ -610,6 +702,9 @@ class AudioToTextConverter:
|
|
|
610
702
|
"fallback_to_model",
|
|
611
703
|
"fallback_reason",
|
|
612
704
|
"fallback_temperature",
|
|
705
|
+
"prompt_variant",
|
|
706
|
+
"fallback_from_prompt_variant",
|
|
707
|
+
"fallback_to_prompt_variant",
|
|
613
708
|
):
|
|
614
709
|
if key in chunk_results[0]:
|
|
615
710
|
result_dict[key] = chunk_results[0][key]
|
|
@@ -11,6 +11,7 @@ from google.genai import types
|
|
|
11
11
|
from google.api_core import exceptions as google_exceptions
|
|
12
12
|
|
|
13
13
|
from ..prompts.ocr import (
|
|
14
|
+
OCR_TO_MARKDOWN_NON_LITERAL_FALLBACK_PROMPT,
|
|
14
15
|
OCR_TO_MARKDOWN_PROMPT,
|
|
15
16
|
OCR_TO_PLAIN_TEXT_PROMPT,
|
|
16
17
|
build_ocr_prompt,
|
|
@@ -34,6 +35,9 @@ OCR_FALLBACK_SOURCE_PATTERN = os.getenv("OCR_FALLBACK_SOURCE_PATTERN", "flash-li
|
|
|
34
35
|
OCR_FALLBACK_MODEL = os.getenv("OCR_FALLBACK_MODEL", "gemini-3-flash-preview")
|
|
35
36
|
OCR_FALLBACK_TEMPERATURE = float(os.getenv("OCR_FALLBACK_TEMPERATURE", "1.0"))
|
|
36
37
|
OCR_FINAL_FALLBACK_MODEL = os.getenv("OCR_FINAL_FALLBACK_MODEL", "gemini-2.0-flash")
|
|
38
|
+
OCR_PROMPT_VARIANT_DEFAULT = "default"
|
|
39
|
+
OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK = "non_literal_fallback"
|
|
40
|
+
OCR_RETRIABLE_OUTPUT_ERROR_CODES = (996, 997, 999)
|
|
37
41
|
|
|
38
42
|
|
|
39
43
|
def compress_and_convert_image(input_path: str, target_size=1):
|
|
@@ -152,7 +156,8 @@ class DocumentOCRToTextConverter:
|
|
|
152
156
|
def __init__(self, ocr_model="gemini-3.1-flash-lite", ocr_model_provider="google",
|
|
153
157
|
markdown_output=True, llm_api_key=None, target_size=1, temp_dir="temp",
|
|
154
158
|
page_range=None, timeout_minutes: int = None, fallback_stage: int = 0,
|
|
155
|
-
max_output_tokens: int | None = None, include_image_descriptions: bool = False
|
|
159
|
+
max_output_tokens: int | None = None, include_image_descriptions: bool = False,
|
|
160
|
+
prompt_variant: str = OCR_PROMPT_VARIANT_DEFAULT):
|
|
156
161
|
"""
|
|
157
162
|
Initialize the DocumentOCRToTextConverter class with specified OCR model and formatting options.
|
|
158
163
|
|
|
@@ -175,6 +180,8 @@ class DocumentOCRToTextConverter:
|
|
|
175
180
|
include_image_descriptions (bool, optional): If True, OCR prompts include
|
|
176
181
|
brief functional descriptions for meaningful non-text images.
|
|
177
182
|
Defaults to False.
|
|
183
|
+
prompt_variant (str, optional): Prompt variant used by this attempt.
|
|
184
|
+
Defaults to "default".
|
|
178
185
|
|
|
179
186
|
Raises:
|
|
180
187
|
OSError: If temp directory creation fails
|
|
@@ -188,6 +195,7 @@ class DocumentOCRToTextConverter:
|
|
|
188
195
|
self.page_range = page_range
|
|
189
196
|
self.timeout_minutes = timeout_minutes
|
|
190
197
|
self.include_image_descriptions = include_image_descriptions
|
|
198
|
+
self.prompt_variant = prompt_variant
|
|
191
199
|
requested_output_tokens = OCR_MAX_OUTPUT_TOKENS if max_output_tokens is None else max_output_tokens
|
|
192
200
|
self.max_output_tokens = max(requested_output_tokens, OCR_MIN_OUTPUT_TOKENS)
|
|
193
201
|
self.fallback_stage = fallback_stage
|
|
@@ -202,16 +210,31 @@ class DocumentOCRToTextConverter:
|
|
|
202
210
|
tempfile.tempdir = self.temp_dir
|
|
203
211
|
|
|
204
212
|
def _build_prompt_template(self) -> str:
|
|
205
|
-
|
|
213
|
+
if self.markdown_output and self.prompt_variant == OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
|
|
214
|
+
base_prompt = OCR_TO_MARKDOWN_NON_LITERAL_FALLBACK_PROMPT
|
|
215
|
+
elif self.markdown_output:
|
|
216
|
+
base_prompt = OCR_TO_MARKDOWN_PROMPT
|
|
217
|
+
else:
|
|
218
|
+
base_prompt = OCR_TO_PLAIN_TEXT_PROMPT
|
|
206
219
|
return build_ocr_prompt(
|
|
207
220
|
base_prompt,
|
|
208
221
|
include_image_descriptions=self.include_image_descriptions,
|
|
209
222
|
)
|
|
210
223
|
|
|
211
|
-
def
|
|
224
|
+
def should_prompt_fallback_retry(self, error: EmptyDocument) -> bool:
|
|
212
225
|
if self.fallback_stage != 0:
|
|
213
226
|
return False
|
|
214
|
-
if
|
|
227
|
+
if not self.markdown_output:
|
|
228
|
+
return False
|
|
229
|
+
if self.prompt_variant == OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK:
|
|
230
|
+
return False
|
|
231
|
+
return error.code in OCR_RETRIABLE_OUTPUT_ERROR_CODES
|
|
232
|
+
|
|
233
|
+
def should_fallback_temperature_retry(self, error: EmptyDocument, temperature: float) -> bool:
|
|
234
|
+
expected_stage = 1 if self.markdown_output else 0
|
|
235
|
+
if self.fallback_stage != expected_stage:
|
|
236
|
+
return False
|
|
237
|
+
if error.code not in OCR_RETRIABLE_OUTPUT_ERROR_CODES:
|
|
215
238
|
return False
|
|
216
239
|
if self.fallback_model == self.ocr_model and temperature == self.fallback_temperature:
|
|
217
240
|
return False
|
|
@@ -220,9 +243,10 @@ class DocumentOCRToTextConverter:
|
|
|
220
243
|
return self.ocr_model != self.fallback_model
|
|
221
244
|
|
|
222
245
|
def should_final_fallback_model(self, error: EmptyDocument) -> bool:
|
|
223
|
-
if self.
|
|
246
|
+
expected_stage = 2 if self.markdown_output else 1
|
|
247
|
+
if self.fallback_stage != expected_stage:
|
|
224
248
|
return False
|
|
225
|
-
if error.code not in
|
|
249
|
+
if error.code not in OCR_RETRIABLE_OUTPUT_ERROR_CODES:
|
|
226
250
|
return False
|
|
227
251
|
if self.final_fallback_model == self.ocr_model:
|
|
228
252
|
return False
|
|
@@ -235,10 +259,13 @@ class DocumentOCRToTextConverter:
|
|
|
235
259
|
fallback_model: str,
|
|
236
260
|
fallback_temperature: float,
|
|
237
261
|
fallback_stage: int,
|
|
262
|
+
prompt_variant: str | None = None,
|
|
238
263
|
) -> dict:
|
|
264
|
+
resolved_prompt_variant = prompt_variant or self.prompt_variant
|
|
239
265
|
logger.info(
|
|
240
|
-
"Retrying document OCR with fallback model %s and temperature %s for %s because %s",
|
|
266
|
+
"Retrying document OCR with fallback model %s, prompt variant %s and temperature %s for %s because %s",
|
|
241
267
|
fallback_model,
|
|
268
|
+
resolved_prompt_variant,
|
|
242
269
|
fallback_temperature,
|
|
243
270
|
file_for_ocr,
|
|
244
271
|
reason,
|
|
@@ -255,6 +282,7 @@ class DocumentOCRToTextConverter:
|
|
|
255
282
|
fallback_stage=fallback_stage,
|
|
256
283
|
max_output_tokens=self.max_output_tokens,
|
|
257
284
|
include_image_descriptions=self.include_image_descriptions,
|
|
285
|
+
prompt_variant=resolved_prompt_variant,
|
|
258
286
|
)
|
|
259
287
|
result = fallback_converter.get_ocr(
|
|
260
288
|
file_for_ocr=file_for_ocr,
|
|
@@ -264,6 +292,8 @@ class DocumentOCRToTextConverter:
|
|
|
264
292
|
result.setdefault("fallback_to_model", fallback_model)
|
|
265
293
|
result.setdefault("fallback_reason", reason)
|
|
266
294
|
result.setdefault("fallback_temperature", fallback_temperature)
|
|
295
|
+
result.setdefault("fallback_from_prompt_variant", self.prompt_variant)
|
|
296
|
+
result.setdefault("fallback_to_prompt_variant", resolved_prompt_variant)
|
|
267
297
|
return result
|
|
268
298
|
|
|
269
299
|
@retry(
|
|
@@ -448,18 +478,28 @@ class DocumentOCRToTextConverter:
|
|
|
448
478
|
"finish_reason": finish_reason,
|
|
449
479
|
"max_output_tokens": self.max_output_tokens,
|
|
450
480
|
"temperature": temperature,
|
|
481
|
+
"prompt_variant": self.prompt_variant,
|
|
451
482
|
}
|
|
452
483
|
|
|
453
484
|
logger.info(f"OCR performed using {self.ocr_model} in {time_elapsed:.2f} seconds")
|
|
454
485
|
return final_ocr_dict
|
|
455
486
|
except EmptyDocument as e:
|
|
487
|
+
if self.should_prompt_fallback_retry(e):
|
|
488
|
+
return self.run_fallback(
|
|
489
|
+
file_for_ocr=file_for_ocr,
|
|
490
|
+
reason=e.message,
|
|
491
|
+
fallback_model=self.ocr_model,
|
|
492
|
+
fallback_temperature=temperature,
|
|
493
|
+
fallback_stage=1,
|
|
494
|
+
prompt_variant=OCR_PROMPT_VARIANT_NON_LITERAL_FALLBACK,
|
|
495
|
+
)
|
|
456
496
|
if self.should_fallback_temperature_retry(e, temperature):
|
|
457
497
|
return self.run_fallback(
|
|
458
498
|
file_for_ocr=file_for_ocr,
|
|
459
499
|
reason=e.message,
|
|
460
500
|
fallback_model=self.fallback_model,
|
|
461
501
|
fallback_temperature=self.fallback_temperature,
|
|
462
|
-
fallback_stage=1,
|
|
502
|
+
fallback_stage=2 if self.markdown_output else 1,
|
|
463
503
|
)
|
|
464
504
|
if self.should_final_fallback_model(e):
|
|
465
505
|
return self.run_fallback(
|
|
@@ -467,7 +507,7 @@ class DocumentOCRToTextConverter:
|
|
|
467
507
|
reason=e.message,
|
|
468
508
|
fallback_model=self.final_fallback_model,
|
|
469
509
|
fallback_temperature=0.0,
|
|
470
|
-
fallback_stage=2,
|
|
510
|
+
fallback_stage=3 if self.markdown_output else 2,
|
|
471
511
|
)
|
|
472
512
|
raise
|
|
473
513
|
|