polytext 0.2.8b3__tar.gz → 0.2.8b4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {polytext-0.2.8b3 → polytext-0.2.8b4}/PKG-INFO +2 -2
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/audio_to_text.py +213 -19
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/beautiful_text.py +63 -14
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/document_ocr_to_text.py +5 -1
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/gemini_quality_guards.py +8 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/ocr_to_text.py +2 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/text_to_md.py +3 -1
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/video_to_audio.py +4 -6
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/base.py +0 -1
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/youtube_llm.py +2 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/processor/audio_chunker.py +4 -5
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/processor/text_merger.py +56 -30
- polytext-0.2.8b4/polytext/prompts/beautiful_text.py +97 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/prompts/transcription.py +32 -12
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext.egg-info/PKG-INFO +2 -2
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext.egg-info/requires.txt +1 -1
- {polytext-0.2.8b3 → polytext-0.2.8b4}/setup.py +1 -1
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_audio_chunker.py +8 -2
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_audio_transcription_model_migration.py +333 -22
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_ocr_fallbacks.py +32 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_ocr_image_descriptions.py +13 -0
- polytext-0.2.8b3/polytext/prompts/beautiful_text.py +0 -61
- {polytext-0.2.8b3 → polytext-0.2.8b4}/LICENSE +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/README.md +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/__init__.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/__init__.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/base.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/document_ocr_to_text_azure_oai.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/html_to_md.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/md_to_text.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/ocr_to_text_azure_oai.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/converter/pdf.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/exceptions/__init__.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/exceptions/base.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/generator/__init__.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/generator/pdf.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/__init__.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/audio.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/aws_auth.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/document.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/document_ocr.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/downloader/__init__.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/downloader/downloader.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/html.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/markdown.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/notebook.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/ocr.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/plain_text.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/video.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/xml_xbrl.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/loader/youtube.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/processor/__init__.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/processor/transcript_chunker.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/prompts/__init__.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/prompts/ocr.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/prompts/text_merging.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/prompts/text_to_md.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/utils/__init__.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext/utils/utils.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext.egg-info/SOURCES.txt +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext.egg-info/dependency_links.txt +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext.egg-info/not-zip-safe +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/polytext.egg-info/top_level.txt +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/pyproject.toml +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/setup.cfg +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_audio_comparison_helpers.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_aws_auth.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_base_loader_error_mapping.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_beautiful_text_manual.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_compare_audio_models.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_compare_document_ocr_to_text_models.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_compare_ocr_to_text_models.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_compare_youtube_models.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_dowload_audio_from_youtube.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_dowload_audio_from_youtube_helpers.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_extracted_text_whitespace.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_gemini_quality_guards.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_audio_transcript_from_gcs.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_customized_pdf_from_markdown.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_document_ocr.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_document_ocr_azure_oai.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_document_text.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_document_text_from_gcs.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_ocr_from_image.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_text_from_markdown.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_get_video_transcript_from_gcs.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_library.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_markdown_loader_gzip.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_markitdown_html.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_notebook_loader.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_pain_text.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_pdf_conversion_error.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_python_version_metadata.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_split_audio_with_llm.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_transcribe_s3_images_from_csv.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_transcribe_s3_images_from_csv_script.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_xml_xbrl_loader.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_youtube_gemini_minimal_check.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_youtube_llm_fallbacks.py +0 -0
- {polytext-0.2.8b3 → polytext-0.2.8b4}/tests/test_youtube_transcript.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: polytext
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.8b4
|
|
4
4
|
Summary: Python utilities to simplify document files management
|
|
5
5
|
Home-page: https://github.com/docsity/polytext
|
|
6
6
|
Author: Matteo Senardi
|
|
@@ -25,7 +25,7 @@ Requires-Dist: markdown-to-json==2.1.2
|
|
|
25
25
|
Requires-Dist: python-docx==1.1.2
|
|
26
26
|
Requires-Dist: google-api-core>=2.24.2
|
|
27
27
|
Requires-Dist: google-cloud-storage<3.0.0,>=2.17
|
|
28
|
-
Requires-Dist: google-genai
|
|
28
|
+
Requires-Dist: google-genai==2.22.0
|
|
29
29
|
Requires-Dist: openai==2.26.0
|
|
30
30
|
Requires-Dist: boto3>=1.42.64
|
|
31
31
|
Requires-Dist: botocore>=1.42.64
|
|
@@ -14,6 +14,7 @@ from google.genai import types
|
|
|
14
14
|
from google.genai import errors as genai_errors
|
|
15
15
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
16
16
|
from google.api_core import exceptions as google_exceptions
|
|
17
|
+
from pydub import AudioSegment
|
|
17
18
|
|
|
18
19
|
from ..exceptions import EmptyDocument
|
|
19
20
|
from ..prompts.transcription import (
|
|
@@ -25,13 +26,22 @@ from ..prompts.transcription import (
|
|
|
25
26
|
)
|
|
26
27
|
from ..processor.audio_chunker import AudioChunker
|
|
27
28
|
from ..processor.text_merger import TextMerger
|
|
28
|
-
from .gemini_quality_guards import
|
|
29
|
+
from .gemini_quality_guards import (
|
|
30
|
+
extract_finish_reason,
|
|
31
|
+
has_excessive_consecutive_word_repetition,
|
|
32
|
+
tail_has_excessive_repetition,
|
|
33
|
+
)
|
|
29
34
|
|
|
30
35
|
logger = logging.getLogger(__name__)
|
|
31
36
|
|
|
32
37
|
SUPPORTED_MIME_TYPES = {
|
|
33
38
|
'audio/x-aac', 'audio/flac', 'audio/mp3', 'audio/m4a', 'audio/mpeg',
|
|
34
|
-
'audio/mpga', 'audio/mp4', 'audio/opus', 'audio/pcm', 'audio/wav', 'audio/webm'
|
|
39
|
+
'audio/mpga', 'audio/mp4', 'audio/opus', 'audio/pcm', 'audio/wav', 'audio/x-wav', 'audio/webm'
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
GEMINI_AUDIO_MIME_ALIASES = {
|
|
43
|
+
'audio/x-aac': 'audio/aac',
|
|
44
|
+
'audio/x-wav': 'audio/wav',
|
|
35
45
|
}
|
|
36
46
|
|
|
37
47
|
INJECTION_GUARD_SYSTEM_INSTRUCTION = (
|
|
@@ -48,16 +58,20 @@ INJECTION_GUARD_SYSTEM_INSTRUCTION = (
|
|
|
48
58
|
)
|
|
49
59
|
|
|
50
60
|
AUDIO_MIN_OUTPUT_TOKENS = 500
|
|
61
|
+
AUDIO_DEFAULT_MAX_OUTPUT_TOKENS = 4096
|
|
62
|
+
AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH = 1
|
|
63
|
+
AUDIO_ADAPTIVE_SPLIT_OVERLAP_MS = 2000
|
|
64
|
+
AUDIO_LONG_DURATION_THRESHOLD_MS = 80 * 60 * 1000
|
|
51
65
|
AUDIO_TAIL_REPETITION_LINES = int(os.getenv("AUDIO_TAIL_REPETITION_LINES", "200"))
|
|
52
66
|
AUDIO_TAIL_REPETITION_THRESHOLD = float(os.getenv("AUDIO_TAIL_REPETITION_THRESHOLD", "0.35"))
|
|
53
67
|
AUDIO_FALLBACK_SOURCE_PATTERN = os.getenv("AUDIO_FALLBACK_SOURCE_PATTERN", "flash-lite")
|
|
54
|
-
AUDIO_FALLBACK_MODEL = os.getenv("AUDIO_FALLBACK_MODEL", "gemini-3-flash-
|
|
68
|
+
AUDIO_FALLBACK_MODEL = os.getenv("AUDIO_FALLBACK_MODEL", "gemini-3.5-flash-lite")
|
|
55
69
|
AUDIO_FALLBACK_TEMPERATURE = float(os.getenv("AUDIO_FALLBACK_TEMPERATURE", "1.0"))
|
|
56
70
|
AUDIO_FINAL_FALLBACK_MODEL = os.getenv("AUDIO_FINAL_FALLBACK_MODEL", "gemini-3.5-flash")
|
|
57
71
|
AUDIO_FILE_UPLOAD_THRESHOLD_BYTES = 20 * 1024 * 1024
|
|
58
72
|
AUDIO_PROMPT_VARIANT_DEFAULT = "default"
|
|
59
73
|
AUDIO_PROMPT_VARIANT_NON_LITERAL_FALLBACK = "non_literal_fallback"
|
|
60
|
-
AUDIO_RETRIABLE_OUTPUT_ERROR_CODES = (996, 997, 999)
|
|
74
|
+
AUDIO_RETRIABLE_OUTPUT_ERROR_CODES = (996, 997, 998, 999)
|
|
61
75
|
NO_HUMAN_SPEECH_MARKER = "no human speech detected"
|
|
62
76
|
|
|
63
77
|
|
|
@@ -123,33 +137,32 @@ def create_ascii_safe_upload_copy(audio_file: str) -> tuple[str, str | None]:
|
|
|
123
137
|
|
|
124
138
|
def compress_and_convert_audio(input_path: str, bitrate_quality: int = 9) -> str:
|
|
125
139
|
"""
|
|
126
|
-
|
|
140
|
+
Normalize an audio file to lossless 16 kHz mono WAV using ffmpeg.
|
|
127
141
|
|
|
128
142
|
Args:
|
|
129
143
|
input_path (str): Path to the original audio file
|
|
130
|
-
bitrate_quality (int, optional):
|
|
144
|
+
bitrate_quality (int, optional): Retained for backward compatibility; WAV output is lossless.
|
|
131
145
|
|
|
132
146
|
Returns:
|
|
133
|
-
str: Path to the temporary
|
|
147
|
+
str: Path to the temporary normalized WAV file
|
|
134
148
|
|
|
135
149
|
Raises:
|
|
136
150
|
RuntimeError: If FFmpeg compression/conversion fails
|
|
137
151
|
|
|
138
152
|
Notes:
|
|
139
|
-
- Creates a temporary
|
|
140
|
-
- Converts audio to mono
|
|
153
|
+
- Creates a temporary WAV file that should be deleted after use
|
|
154
|
+
- Converts audio to 16-bit PCM mono at 16kHz
|
|
141
155
|
- Uses maximum available CPU threads for faster processing
|
|
142
156
|
"""
|
|
143
157
|
# Create temporary file for audio output
|
|
144
|
-
fd, temp_audio_path = tempfile.mkstemp(suffix='.
|
|
158
|
+
fd, temp_audio_path = tempfile.mkstemp(suffix='.wav')
|
|
145
159
|
os.close(fd)
|
|
146
160
|
|
|
147
|
-
logger.info(
|
|
161
|
+
logger.info("Normalizing audio to lossless 16 kHz mono WAV")
|
|
148
162
|
try:
|
|
149
163
|
ffmpeg.input(input_path).output(
|
|
150
164
|
temp_audio_path,
|
|
151
|
-
|
|
152
|
-
acodec='libmp3lame',
|
|
165
|
+
acodec='pcm_s16le',
|
|
153
166
|
ac=1, # Convert to mono
|
|
154
167
|
ar=16000, # Lower sample rate
|
|
155
168
|
vn=None,
|
|
@@ -162,7 +175,7 @@ def compress_and_convert_audio(input_path: str, bitrate_quality: int = 9) -> str
|
|
|
162
175
|
os.unlink(temp_audio_path)
|
|
163
176
|
raise
|
|
164
177
|
|
|
165
|
-
logger.info(f"Successfully
|
|
178
|
+
logger.info(f"Successfully normalized audio: {temp_audio_path}")
|
|
166
179
|
return temp_audio_path
|
|
167
180
|
|
|
168
181
|
|
|
@@ -189,7 +202,7 @@ def transcribe_full_audio(audio_file, markdown_output: bool = False,
|
|
|
189
202
|
bitrate_quality (int, optional): Variable bitrate quality from 0-9 (9 being lowest). Defaults to 9
|
|
190
203
|
timeout_minutes (int, optional): Number of minutes to wait for a response. Defaults to None.
|
|
191
204
|
max_llm_tokens (int, optional): Token budget used for audio chunk sizing. Defaults to 4250.
|
|
192
|
-
max_output_tokens (int | None, optional): Maximum Gemini output tokens. Defaults to
|
|
205
|
+
max_output_tokens (int | None, optional): Maximum Gemini output tokens. Defaults to 4096.
|
|
193
206
|
is_output_audio_raw (bool, optional): If True, use the raw Markdown audio prompt.
|
|
194
207
|
If False, use the formatted Markdown audio prompt. Defaults to True.
|
|
195
208
|
|
|
@@ -211,6 +224,8 @@ class AudioToTextConverter:
|
|
|
211
224
|
max_output_tokens: int | None = None, temp_dir: str = "temp",
|
|
212
225
|
bitrate_quality: int = 9, timeout_minutes: int = None,
|
|
213
226
|
fallback_stage: int = 0,
|
|
227
|
+
adaptive_split_depth: int = 0,
|
|
228
|
+
long_audio_protections_enabled: bool = False,
|
|
214
229
|
prompt_variant: str = AUDIO_PROMPT_VARIANT_DEFAULT,
|
|
215
230
|
is_output_audio_raw: bool = True):
|
|
216
231
|
"""
|
|
@@ -225,12 +240,14 @@ class AudioToTextConverter:
|
|
|
225
240
|
llm_api_key (str, optional): Override API key for language model. Defaults to None.
|
|
226
241
|
max_llm_tokens (int): Token budget used to size audio chunks. Defaults to 4250.
|
|
227
242
|
max_output_tokens (int | None): Maximum number of output tokens for Gemini generation.
|
|
228
|
-
Defaults to
|
|
243
|
+
Defaults to 4096.
|
|
229
244
|
temp_dir (str): Directory for temporary files. Defaults to "temp".
|
|
230
245
|
bitrate_quality (int, optional): Variable bitrate quality from 0-9 (9 being lowest). Defaults to 9
|
|
231
246
|
timeout_minutes (int): Number of minutes to wait for a response.
|
|
232
247
|
fallback_stage (int, optional): Internal retry stage used by fallback attempts.
|
|
233
248
|
Defaults to 0.
|
|
249
|
+
long_audio_protections_enabled (bool, optional): Apply stricter recovery rules inherited
|
|
250
|
+
from an original audio longer than 80 minutes. Defaults to False.
|
|
234
251
|
prompt_variant (str, optional): Prompt variant used by this attempt.
|
|
235
252
|
Defaults to "default".
|
|
236
253
|
is_output_audio_raw (bool, optional): If True, use the raw Markdown audio prompt.
|
|
@@ -247,12 +264,18 @@ class AudioToTextConverter:
|
|
|
247
264
|
self.markdown_output = markdown_output
|
|
248
265
|
self.llm_api_key = llm_api_key
|
|
249
266
|
self.max_llm_tokens = max(max_llm_tokens, AUDIO_MIN_OUTPUT_TOKENS)
|
|
250
|
-
requested_output_tokens =
|
|
267
|
+
requested_output_tokens = (
|
|
268
|
+
AUDIO_DEFAULT_MAX_OUTPUT_TOKENS
|
|
269
|
+
if max_output_tokens is None
|
|
270
|
+
else max_output_tokens
|
|
271
|
+
)
|
|
251
272
|
self.max_output_tokens = max(requested_output_tokens, AUDIO_MIN_OUTPUT_TOKENS)
|
|
252
273
|
self.chunked_audio = False
|
|
253
274
|
self.bitrate_quality = bitrate_quality
|
|
254
275
|
self.timeout_minutes = timeout_minutes
|
|
255
276
|
self.fallback_stage = fallback_stage
|
|
277
|
+
self.adaptive_split_depth = adaptive_split_depth
|
|
278
|
+
self.long_audio_protections_enabled = long_audio_protections_enabled
|
|
256
279
|
self.prompt_variant = prompt_variant
|
|
257
280
|
self.fallback_source_pattern = AUDIO_FALLBACK_SOURCE_PATTERN
|
|
258
281
|
self.fallback_model = AUDIO_FALLBACK_MODEL
|
|
@@ -276,6 +299,9 @@ class AudioToTextConverter:
|
|
|
276
299
|
return AUDIO_TO_MARKDOWN_PROMPT
|
|
277
300
|
return AUDIO_TO_PLAIN_TEXT_PROMPT
|
|
278
301
|
|
|
302
|
+
def set_long_audio_protections(self, duration_ms: int) -> None:
|
|
303
|
+
self.long_audio_protections_enabled = duration_ms > AUDIO_LONG_DURATION_THRESHOLD_MS
|
|
304
|
+
|
|
279
305
|
def should_prompt_fallback_retry(self, error: EmptyDocument) -> bool:
|
|
280
306
|
if self.fallback_stage != 0:
|
|
281
307
|
return False
|
|
@@ -338,6 +364,8 @@ class AudioToTextConverter:
|
|
|
338
364
|
bitrate_quality=self.bitrate_quality,
|
|
339
365
|
timeout_minutes=self.timeout_minutes,
|
|
340
366
|
fallback_stage=fallback_stage,
|
|
367
|
+
adaptive_split_depth=self.adaptive_split_depth,
|
|
368
|
+
long_audio_protections_enabled=self.long_audio_protections_enabled,
|
|
341
369
|
prompt_variant=resolved_prompt_variant,
|
|
342
370
|
is_output_audio_raw=self.is_output_audio_raw,
|
|
343
371
|
)
|
|
@@ -353,10 +381,83 @@ class AudioToTextConverter:
|
|
|
353
381
|
result.setdefault("fallback_to_prompt_variant", resolved_prompt_variant)
|
|
354
382
|
return result
|
|
355
383
|
|
|
384
|
+
def transcribe_audio_halves(self, audio_file: str, temperature: float = 0.0) -> dict:
|
|
385
|
+
"""Split one genuinely overlong chunk and transcribe both halves once."""
|
|
386
|
+
audio = AudioSegment.from_file(audio_file)
|
|
387
|
+
midpoint = len(audio) // 2
|
|
388
|
+
ranges = (
|
|
389
|
+
(0, min(len(audio), midpoint + AUDIO_ADAPTIVE_SPLIT_OVERLAP_MS)),
|
|
390
|
+
(max(0, midpoint - AUDIO_ADAPTIVE_SPLIT_OVERLAP_MS), len(audio)),
|
|
391
|
+
)
|
|
392
|
+
split_paths = []
|
|
393
|
+
split_results = []
|
|
394
|
+
|
|
395
|
+
try:
|
|
396
|
+
for start_ms, end_ms in ranges:
|
|
397
|
+
fd, split_path = tempfile.mkstemp(
|
|
398
|
+
prefix="adaptive-audio-split-",
|
|
399
|
+
suffix=".wav",
|
|
400
|
+
dir=self.temp_dir,
|
|
401
|
+
)
|
|
402
|
+
os.close(fd)
|
|
403
|
+
split_paths.append(split_path)
|
|
404
|
+
(
|
|
405
|
+
audio[start_ms:end_ms]
|
|
406
|
+
.set_frame_rate(16000)
|
|
407
|
+
.set_channels(1)
|
|
408
|
+
.set_sample_width(2)
|
|
409
|
+
.export(split_path, format="wav", codec="pcm_s16le")
|
|
410
|
+
.close()
|
|
411
|
+
)
|
|
412
|
+
|
|
413
|
+
split_converter = AudioToTextConverter(
|
|
414
|
+
transcription_model=self.transcription_model,
|
|
415
|
+
transcription_model_provider=self.transcription_model_provider,
|
|
416
|
+
k=self.k,
|
|
417
|
+
min_matches=self.min_matches,
|
|
418
|
+
markdown_output=self.markdown_output,
|
|
419
|
+
llm_api_key=self.llm_api_key,
|
|
420
|
+
max_llm_tokens=self.max_llm_tokens,
|
|
421
|
+
max_output_tokens=self.max_output_tokens,
|
|
422
|
+
temp_dir=self.temp_dir,
|
|
423
|
+
bitrate_quality=self.bitrate_quality,
|
|
424
|
+
timeout_minutes=self.timeout_minutes,
|
|
425
|
+
fallback_stage=self.fallback_stage,
|
|
426
|
+
adaptive_split_depth=self.adaptive_split_depth + 1,
|
|
427
|
+
long_audio_protections_enabled=self.long_audio_protections_enabled,
|
|
428
|
+
prompt_variant=self.prompt_variant,
|
|
429
|
+
is_output_audio_raw=self.is_output_audio_raw,
|
|
430
|
+
)
|
|
431
|
+
split_results.append(
|
|
432
|
+
split_converter.transcribe_audio(split_path, temperature=temperature)
|
|
433
|
+
)
|
|
434
|
+
|
|
435
|
+
merged_transcript = TextMerger(k=self.k, min_matches=self.min_matches).merge_texts(
|
|
436
|
+
split_results[0]["transcript"],
|
|
437
|
+
split_results[1]["transcript"],
|
|
438
|
+
)
|
|
439
|
+
return {
|
|
440
|
+
"transcript": merged_transcript,
|
|
441
|
+
"completion_tokens": sum(item["completion_tokens"] for item in split_results),
|
|
442
|
+
"prompt_tokens": sum(item["prompt_tokens"] for item in split_results),
|
|
443
|
+
"completion_model": self.transcription_model,
|
|
444
|
+
"completion_model_provider": self.transcription_model_provider,
|
|
445
|
+
"finish_reason": "ADAPTIVE_SPLIT",
|
|
446
|
+
"max_output_tokens": self.max_output_tokens,
|
|
447
|
+
"temperature": temperature,
|
|
448
|
+
"prompt_variant": self.prompt_variant,
|
|
449
|
+
"adaptive_split": True,
|
|
450
|
+
"split_results": split_results,
|
|
451
|
+
}
|
|
452
|
+
finally:
|
|
453
|
+
for split_path in split_paths:
|
|
454
|
+
if os.path.exists(split_path):
|
|
455
|
+
os.remove(split_path)
|
|
456
|
+
|
|
356
457
|
def build_config(self, output_budget: int, temperature: float = 0.0) -> types.GenerateContentConfig:
|
|
357
458
|
return types.GenerateContentConfig(
|
|
358
459
|
temperature=temperature,
|
|
359
|
-
thinking_config=types.ThinkingConfig(
|
|
460
|
+
thinking_config=types.ThinkingConfig(thinking_level="minimal"),
|
|
360
461
|
max_output_tokens=output_budget,
|
|
361
462
|
system_instruction=INJECTION_GUARD_SYSTEM_INSTRUCTION,
|
|
362
463
|
tools=[],
|
|
@@ -433,6 +534,7 @@ class AudioToTextConverter:
|
|
|
433
534
|
except ValueError:
|
|
434
535
|
logger.exception("Unsupported audio format for %s", audio_file)
|
|
435
536
|
raise
|
|
537
|
+
mime_type = GEMINI_AUDIO_MIME_ALIASES.get(mime_type, mime_type)
|
|
436
538
|
|
|
437
539
|
return client.models.generate_content(
|
|
438
540
|
model=self.transcription_model,
|
|
@@ -455,7 +557,6 @@ class AudioToTextConverter:
|
|
|
455
557
|
google_exceptions.ServiceUnavailable,
|
|
456
558
|
google_exceptions.InternalServerError,
|
|
457
559
|
genai_errors.ServerError,
|
|
458
|
-
genai_errors.APIError,
|
|
459
560
|
),
|
|
460
561
|
tries=8,
|
|
461
562
|
delay=1,
|
|
@@ -518,6 +619,9 @@ class AudioToTextConverter:
|
|
|
518
619
|
tail_lines=AUDIO_TAIL_REPETITION_LINES,
|
|
519
620
|
threshold=AUDIO_TAIL_REPETITION_THRESHOLD,
|
|
520
621
|
)
|
|
622
|
+
has_repetitive_word_loop = has_excessive_consecutive_word_repetition(
|
|
623
|
+
response_text,
|
|
624
|
+
)
|
|
521
625
|
usage_metadata = getattr(response, "usage_metadata", None)
|
|
522
626
|
completion_tokens = getattr(usage_metadata, "candidates_token_count", 0) or 0
|
|
523
627
|
prompt_tokens = getattr(usage_metadata, "prompt_token_count", 0) or 0
|
|
@@ -532,19 +636,84 @@ class AudioToTextConverter:
|
|
|
532
636
|
)
|
|
533
637
|
|
|
534
638
|
if finish_reason and "MAX_TOKENS" in finish_reason:
|
|
639
|
+
if (
|
|
640
|
+
self.long_audio_protections_enabled
|
|
641
|
+
and self.adaptive_split_depth < AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH
|
|
642
|
+
):
|
|
643
|
+
logger.info(
|
|
644
|
+
"Splitting long-audio chunk before fallback after MAX_TOKENS response: %s",
|
|
645
|
+
audio_file,
|
|
646
|
+
)
|
|
647
|
+
return self.transcribe_audio_halves(audio_file, temperature=temperature)
|
|
648
|
+
if has_repetitive_tail or has_repetitive_word_loop:
|
|
649
|
+
raise EmptyDocument(
|
|
650
|
+
message=f"Transcript discarded because repetitive output reached max tokens for audio: {audio_file}",
|
|
651
|
+
code=997,
|
|
652
|
+
)
|
|
653
|
+
if self.adaptive_split_depth < AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH:
|
|
654
|
+
logger.info(
|
|
655
|
+
"Splitting audio chunk after non-repetitive MAX_TOKENS response: %s",
|
|
656
|
+
audio_file,
|
|
657
|
+
)
|
|
658
|
+
return self.transcribe_audio_halves(audio_file, temperature=temperature)
|
|
535
659
|
raise EmptyDocument(
|
|
536
660
|
message=f"Transcript truncated because max output tokens were reached for audio: {audio_file}",
|
|
537
661
|
code=999,
|
|
538
662
|
)
|
|
539
663
|
|
|
540
664
|
if has_repetitive_tail:
|
|
665
|
+
if (
|
|
666
|
+
self.long_audio_protections_enabled
|
|
667
|
+
and self.adaptive_split_depth < AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH
|
|
668
|
+
):
|
|
669
|
+
logger.info(
|
|
670
|
+
"Splitting long-audio chunk before fallback after repetitive tail: %s",
|
|
671
|
+
audio_file,
|
|
672
|
+
)
|
|
673
|
+
return self.transcribe_audio_halves(audio_file, temperature=temperature)
|
|
541
674
|
raise EmptyDocument(
|
|
542
675
|
message=f"Transcript discarded because repetitive tail was detected for audio: {audio_file}",
|
|
543
676
|
code=997,
|
|
544
677
|
)
|
|
545
678
|
|
|
679
|
+
if has_repetitive_word_loop:
|
|
680
|
+
if (
|
|
681
|
+
self.long_audio_protections_enabled
|
|
682
|
+
and self.adaptive_split_depth < AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH
|
|
683
|
+
):
|
|
684
|
+
logger.info(
|
|
685
|
+
"Splitting long-audio chunk before fallback after repetitive word loop: %s",
|
|
686
|
+
audio_file,
|
|
687
|
+
)
|
|
688
|
+
return self.transcribe_audio_halves(audio_file, temperature=temperature)
|
|
689
|
+
raise EmptyDocument(
|
|
690
|
+
message=f"Transcript discarded because repetitive word loop was detected for audio: {audio_file}",
|
|
691
|
+
code=997,
|
|
692
|
+
)
|
|
693
|
+
|
|
546
694
|
response_text, marker_only = normalize_no_human_speech_marker(response_text)
|
|
547
695
|
|
|
696
|
+
is_insignificant_stop = (
|
|
697
|
+
finish_reason
|
|
698
|
+
and "STOP" in finish_reason
|
|
699
|
+
and (
|
|
700
|
+
not response_text.strip()
|
|
701
|
+
or (
|
|
702
|
+
completion_tokens <= 4
|
|
703
|
+
and len(response_text.split()) <= 2
|
|
704
|
+
)
|
|
705
|
+
)
|
|
706
|
+
)
|
|
707
|
+
if (
|
|
708
|
+
self.long_audio_protections_enabled
|
|
709
|
+
and not marker_only
|
|
710
|
+
and is_insignificant_stop
|
|
711
|
+
):
|
|
712
|
+
raise EmptyDocument(
|
|
713
|
+
message=f"Transcript discarded because STOP returned empty or insignificant output for audio: {audio_file}",
|
|
714
|
+
code=998,
|
|
715
|
+
)
|
|
716
|
+
|
|
548
717
|
response_dict = {
|
|
549
718
|
"transcript": "" if marker_only else response_text,
|
|
550
719
|
"completion_tokens": completion_tokens,
|
|
@@ -562,6 +731,24 @@ class AudioToTextConverter:
|
|
|
562
731
|
)
|
|
563
732
|
return response_dict
|
|
564
733
|
except EmptyDocument as e:
|
|
734
|
+
if (
|
|
735
|
+
e.code == 999
|
|
736
|
+
and self.adaptive_split_depth >= AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH
|
|
737
|
+
and self.fallback_stage == 0
|
|
738
|
+
and self.transcription_model != self.fallback_model
|
|
739
|
+
):
|
|
740
|
+
return self.run_fallback(
|
|
741
|
+
audio_file=audio_file,
|
|
742
|
+
reason=e.message,
|
|
743
|
+
fallback_model=self.fallback_model,
|
|
744
|
+
fallback_temperature=self.fallback_temperature,
|
|
745
|
+
fallback_stage=2 if self.markdown_output else 1,
|
|
746
|
+
prompt_variant=(
|
|
747
|
+
AUDIO_PROMPT_VARIANT_NON_LITERAL_FALLBACK
|
|
748
|
+
if self.markdown_output
|
|
749
|
+
else self.prompt_variant
|
|
750
|
+
),
|
|
751
|
+
)
|
|
565
752
|
if self.should_prompt_fallback_retry(e):
|
|
566
753
|
return self.run_fallback(
|
|
567
754
|
audio_file=audio_file,
|
|
@@ -654,6 +841,13 @@ class AudioToTextConverter:
|
|
|
654
841
|
# Create chunker and extract chunks
|
|
655
842
|
logger.info("Creating AudioChunker instance...")
|
|
656
843
|
chunker = AudioChunker(used_file, max_llm_tokens=self.max_llm_tokens)
|
|
844
|
+
self.set_long_audio_protections(chunker.duration_ms)
|
|
845
|
+
logger.info(
|
|
846
|
+
"Long-audio transcription protections enabled: %s (duration: %sms, threshold: %sms)",
|
|
847
|
+
self.long_audio_protections_enabled,
|
|
848
|
+
chunker.duration_ms,
|
|
849
|
+
AUDIO_LONG_DURATION_THRESHOLD_MS,
|
|
850
|
+
)
|
|
657
851
|
chunks = chunker.extract_chunks()
|
|
658
852
|
|
|
659
853
|
logger.info(f"chunks: {chunks}")
|
|
@@ -9,8 +9,6 @@ from google.api_core import exceptions as google_exceptions
|
|
|
9
9
|
from retry import retry
|
|
10
10
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
11
11
|
|
|
12
|
-
from polytext.processor.transcript_chunker import TranscriptChunker
|
|
13
|
-
from polytext.processor.text_merger import TextMerger
|
|
14
12
|
from polytext.prompts.beautiful_text import BEAUTIFUL_TEXT_PROMPT
|
|
15
13
|
|
|
16
14
|
logger = logging.getLogger(__name__)
|
|
@@ -26,6 +24,7 @@ class BeautifulTextConverter:
|
|
|
26
24
|
prompt_overhead: int = 1800,
|
|
27
25
|
tokens_per_char: float = 0.25,
|
|
28
26
|
overlap_chars: int = 800,
|
|
27
|
+
max_target_chars: int = 12000,
|
|
29
28
|
) -> None:
|
|
30
29
|
self.llm_api_key = llm_api_key
|
|
31
30
|
self.model = model
|
|
@@ -34,19 +33,58 @@ class BeautifulTextConverter:
|
|
|
34
33
|
self.prompt_overhead = prompt_overhead
|
|
35
34
|
self.tokens_per_char = tokens_per_char
|
|
36
35
|
self.overlap_chars = overlap_chars
|
|
36
|
+
self.max_target_chars = max_target_chars
|
|
37
37
|
|
|
38
38
|
def get_client(self):
|
|
39
39
|
return genai.Client(api_key=self.llm_api_key) if self.llm_api_key else genai.Client()
|
|
40
40
|
|
|
41
41
|
def chunk_raw_text(self, raw_text: str) -> list[dict]:
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
42
|
+
text = (raw_text or "").strip()
|
|
43
|
+
if not text:
|
|
44
|
+
return []
|
|
45
|
+
|
|
46
|
+
chunks = []
|
|
47
|
+
start = 0
|
|
48
|
+
index = 0
|
|
49
|
+
|
|
50
|
+
while start < len(text):
|
|
51
|
+
proposed_end = min(start + self.max_target_chars, len(text))
|
|
52
|
+
end = self._find_chunk_end(text, start, proposed_end)
|
|
53
|
+
target_text = text[start:end].strip()
|
|
54
|
+
|
|
55
|
+
if not target_text:
|
|
56
|
+
break
|
|
57
|
+
|
|
58
|
+
chunks.append(
|
|
59
|
+
{
|
|
60
|
+
"index": index,
|
|
61
|
+
"target_text": target_text,
|
|
62
|
+
}
|
|
63
|
+
)
|
|
64
|
+
start = end
|
|
65
|
+
index += 1
|
|
66
|
+
|
|
67
|
+
return chunks
|
|
68
|
+
|
|
69
|
+
def _find_chunk_end(self, text: str, start: int, proposed_end: int) -> int:
|
|
70
|
+
if proposed_end >= len(text):
|
|
71
|
+
return len(text)
|
|
72
|
+
|
|
73
|
+
minimum_end = start + int(self.max_target_chars * 0.65)
|
|
74
|
+
chunk_window = text[minimum_end:proposed_end]
|
|
75
|
+
sentence_boundaries = list(re.finditer(r"(?<=[.!?])\s+", chunk_window))
|
|
76
|
+
if sentence_boundaries:
|
|
77
|
+
return minimum_end + sentence_boundaries[-1].end()
|
|
78
|
+
|
|
79
|
+
paragraph_boundary = text.rfind("\n\n", minimum_end, proposed_end)
|
|
80
|
+
if paragraph_boundary != -1:
|
|
81
|
+
return paragraph_boundary + 2
|
|
82
|
+
|
|
83
|
+
whitespace_boundary = text.rfind(" ", minimum_end, proposed_end)
|
|
84
|
+
if whitespace_boundary != -1:
|
|
85
|
+
return whitespace_boundary + 1
|
|
86
|
+
|
|
87
|
+
return proposed_end
|
|
50
88
|
|
|
51
89
|
@retry(
|
|
52
90
|
(
|
|
@@ -60,11 +98,18 @@ class BeautifulTextConverter:
|
|
|
60
98
|
backoff=2,
|
|
61
99
|
logger=logger,
|
|
62
100
|
)
|
|
63
|
-
def process_chunk(self, client,
|
|
101
|
+
def process_chunk(self, client, chunk: dict, index: int) -> dict:
|
|
64
102
|
logger.info("Processing beautiful text chunk %s", index + 1)
|
|
65
103
|
start_time = time.time()
|
|
66
104
|
|
|
105
|
+
target_text = chunk["target_text"]
|
|
106
|
+
|
|
67
107
|
config = types.GenerateContentConfig(
|
|
108
|
+
temperature=0,
|
|
109
|
+
thinking_config=types.ThinkingConfig(thinking_budget=0),
|
|
110
|
+
max_output_tokens=self.max_llm_tokens,
|
|
111
|
+
tools=[],
|
|
112
|
+
automatic_function_calling=types.AutomaticFunctionCallingConfig(disable=True),
|
|
68
113
|
safety_settings=[
|
|
69
114
|
types.SafetySetting(
|
|
70
115
|
category=types.HarmCategory.HARM_CATEGORY_HATE_SPEECH,
|
|
@@ -87,7 +132,11 @@ class BeautifulTextConverter:
|
|
|
87
132
|
|
|
88
133
|
response = client.models.generate_content(
|
|
89
134
|
model=self.model,
|
|
90
|
-
contents=[
|
|
135
|
+
contents=[
|
|
136
|
+
BEAUTIFUL_TEXT_PROMPT,
|
|
137
|
+
"TARGET TEXT TO CLEAN",
|
|
138
|
+
target_text,
|
|
139
|
+
],
|
|
91
140
|
config=config,
|
|
92
141
|
)
|
|
93
142
|
|
|
@@ -100,7 +149,7 @@ class BeautifulTextConverter:
|
|
|
100
149
|
}
|
|
101
150
|
|
|
102
151
|
def merge_cleaned_chunks(self, chunks: list[str]) -> str:
|
|
103
|
-
return
|
|
152
|
+
return "\n\n".join(chunk.strip() for chunk in chunks if chunk.strip())
|
|
104
153
|
|
|
105
154
|
def _convert_markdown_to_json(self, markdown_text: str) -> dict:
|
|
106
155
|
if not markdown_text.strip():
|
|
@@ -181,7 +230,7 @@ class BeautifulTextConverter:
|
|
|
181
230
|
|
|
182
231
|
with ThreadPoolExecutor() as executor:
|
|
183
232
|
future_to_index = {
|
|
184
|
-
executor.submit(self.process_chunk, client, chunk
|
|
233
|
+
executor.submit(self.process_chunk, client, chunk, chunk["index"]): chunk["index"]
|
|
185
234
|
for chunk in chunks
|
|
186
235
|
}
|
|
187
236
|
|
|
@@ -242,7 +242,9 @@ class DocumentOCRToTextConverter:
|
|
|
242
242
|
return error.code in OCR_RETRIABLE_OUTPUT_ERROR_CODES
|
|
243
243
|
|
|
244
244
|
def should_fallback_temperature_retry(self, error: EmptyDocument, temperature: float) -> bool:
|
|
245
|
-
|
|
245
|
+
# The non-literal prompt retry advances every OCR mode to stage 1.
|
|
246
|
+
# Plain-text document OCR must therefore also try the fallback model at stage 1.
|
|
247
|
+
expected_stage = 1
|
|
246
248
|
if self.fallback_stage != expected_stage:
|
|
247
249
|
return False
|
|
248
250
|
if error.code not in OCR_RETRIABLE_OUTPUT_ERROR_CODES:
|
|
@@ -368,6 +370,8 @@ class DocumentOCRToTextConverter:
|
|
|
368
370
|
config = types.GenerateContentConfig(
|
|
369
371
|
temperature=temperature,
|
|
370
372
|
max_output_tokens=self.max_output_tokens,
|
|
373
|
+
tools=[],
|
|
374
|
+
automatic_function_calling=types.AutomaticFunctionCallingConfig(disable=True),
|
|
371
375
|
safety_settings=[
|
|
372
376
|
types.SafetySetting(
|
|
373
377
|
category=types.HarmCategory.HARM_CATEGORY_HATE_SPEECH,
|
|
@@ -37,6 +37,14 @@ def has_consecutive_repetition(items: list[str], min_run_length: int = 3) -> boo
|
|
|
37
37
|
return False
|
|
38
38
|
|
|
39
39
|
|
|
40
|
+
def has_excessive_consecutive_word_repetition(
|
|
41
|
+
text: str,
|
|
42
|
+
min_run_length: int = 12,
|
|
43
|
+
) -> bool:
|
|
44
|
+
words = re.findall(r"\b\w+\b", (text or "").casefold())
|
|
45
|
+
return has_consecutive_repetition(words, min_run_length=min_run_length)
|
|
46
|
+
|
|
47
|
+
|
|
40
48
|
def tail_has_excessive_repetition(
|
|
41
49
|
text: str,
|
|
42
50
|
tail_lines: int,
|
|
@@ -352,6 +352,8 @@ class OCRToTextConverter:
|
|
|
352
352
|
config = types.GenerateContentConfig(
|
|
353
353
|
temperature=temperature,
|
|
354
354
|
max_output_tokens=self.max_output_tokens,
|
|
355
|
+
tools=[],
|
|
356
|
+
automatic_function_calling=types.AutomaticFunctionCallingConfig(disable=True),
|
|
355
357
|
safety_settings=[
|
|
356
358
|
types.SafetySetting(
|
|
357
359
|
category=types.HarmCategory.HARM_CATEGORY_HATE_SPEECH,
|
|
@@ -141,6 +141,8 @@ class TextToMdConverter:
|
|
|
141
141
|
start_time = time.time()
|
|
142
142
|
|
|
143
143
|
config = types.GenerateContentConfig(
|
|
144
|
+
tools=[],
|
|
145
|
+
automatic_function_calling=types.AutomaticFunctionCallingConfig(disable=True),
|
|
144
146
|
safety_settings=[
|
|
145
147
|
types.SafetySetting(category=types.HarmCategory.HARM_CATEGORY_HATE_SPEECH, threshold=types.HarmBlockThreshold.BLOCK_NONE),
|
|
146
148
|
types.SafetySetting(category=types.HarmCategory.HARM_CATEGORY_DANGEROUS_CONTENT, threshold=types.HarmBlockThreshold.BLOCK_NONE),
|
|
@@ -235,4 +237,4 @@ class TextToMdConverter:
|
|
|
235
237
|
|
|
236
238
|
logging.info(f"**** FINISH ****")
|
|
237
239
|
|
|
238
|
-
return result_dict
|
|
240
|
+
return result_dict
|
|
@@ -12,7 +12,7 @@ def convert_video_to_audio(video_file: str , bitrate_quality: int =9) -> str:
|
|
|
12
12
|
|
|
13
13
|
Args:
|
|
14
14
|
video_file (str): Path to the video file.
|
|
15
|
-
bitrate_quality (int, optional):
|
|
15
|
+
bitrate_quality (int, optional): Retained for backward compatibility; WAV output is lossless.
|
|
16
16
|
|
|
17
17
|
Returns:
|
|
18
18
|
str: Path to the converted audio file.
|
|
@@ -22,12 +22,12 @@ def convert_video_to_audio(video_file: str , bitrate_quality: int =9) -> str:
|
|
|
22
22
|
Exception: If any other error occurs during conversion
|
|
23
23
|
"""
|
|
24
24
|
|
|
25
|
-
logger.info(
|
|
25
|
+
logger.info("Converting video to lossless 16 kHz mono WAV.")
|
|
26
26
|
|
|
27
27
|
temp_audio_path = None
|
|
28
28
|
try:
|
|
29
29
|
# Create temporary file for audio output
|
|
30
|
-
fd, temp_audio_path = tempfile.mkstemp(suffix='.
|
|
30
|
+
fd, temp_audio_path = tempfile.mkstemp(suffix='.wav')
|
|
31
31
|
os.close(fd)
|
|
32
32
|
|
|
33
33
|
# Simple efficient pipeline
|
|
@@ -35,9 +35,7 @@ def convert_video_to_audio(video_file: str , bitrate_quality: int =9) -> str:
|
|
|
35
35
|
ffmpeg
|
|
36
36
|
.input(video_file)
|
|
37
37
|
.output(temp_audio_path,
|
|
38
|
-
acodec='
|
|
39
|
-
# ab='64k',
|
|
40
|
-
q=bitrate_quality, # Variable bitrate quality (0-9, 9 being lowest)
|
|
38
|
+
acodec='pcm_s16le',
|
|
41
39
|
ac=1, # Convert to mono
|
|
42
40
|
ar=16000, # Lower sample rate
|
|
43
41
|
vn=None,
|