polytext 0.2.8b2__tar.gz → 0.2.8b4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {polytext-0.2.8b2 → polytext-0.2.8b4}/PKG-INFO +2 -2
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/audio_to_text.py +222 -23
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/beautiful_text.py +63 -14
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/document_ocr_to_text.py +5 -1
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/gemini_quality_guards.py +8 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/ocr_to_text.py +2 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/text_to_md.py +3 -1
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/video_to_audio.py +4 -6
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/base.py +0 -1
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/youtube_llm.py +2 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/processor/audio_chunker.py +4 -5
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/processor/text_merger.py +56 -30
- polytext-0.2.8b4/polytext/prompts/beautiful_text.py +97 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/prompts/transcription.py +32 -12
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext.egg-info/PKG-INFO +2 -2
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext.egg-info/requires.txt +1 -1
- {polytext-0.2.8b2 → polytext-0.2.8b4}/setup.py +1 -1
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_audio_chunker.py +8 -2
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_audio_transcription_model_migration.py +333 -22
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_ocr_fallbacks.py +32 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_ocr_image_descriptions.py +13 -0
- polytext-0.2.8b2/polytext/prompts/beautiful_text.py +0 -61
- {polytext-0.2.8b2 → polytext-0.2.8b4}/LICENSE +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/README.md +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/__init__.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/__init__.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/base.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/document_ocr_to_text_azure_oai.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/html_to_md.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/md_to_text.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/ocr_to_text_azure_oai.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/converter/pdf.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/exceptions/__init__.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/exceptions/base.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/generator/__init__.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/generator/pdf.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/__init__.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/audio.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/aws_auth.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/document.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/document_ocr.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/downloader/__init__.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/downloader/downloader.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/html.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/markdown.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/notebook.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/ocr.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/plain_text.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/video.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/xml_xbrl.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/loader/youtube.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/processor/__init__.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/processor/transcript_chunker.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/prompts/__init__.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/prompts/ocr.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/prompts/text_merging.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/prompts/text_to_md.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/utils/__init__.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext/utils/utils.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext.egg-info/SOURCES.txt +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext.egg-info/dependency_links.txt +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext.egg-info/not-zip-safe +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/polytext.egg-info/top_level.txt +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/pyproject.toml +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/setup.cfg +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_audio_comparison_helpers.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_aws_auth.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_base_loader_error_mapping.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_beautiful_text_manual.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_compare_audio_models.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_compare_document_ocr_to_text_models.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_compare_ocr_to_text_models.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_compare_youtube_models.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_dowload_audio_from_youtube.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_dowload_audio_from_youtube_helpers.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_extracted_text_whitespace.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_gemini_quality_guards.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_audio_transcript_from_gcs.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_customized_pdf_from_markdown.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_document_ocr.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_document_ocr_azure_oai.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_document_text.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_document_text_from_gcs.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_ocr_from_image.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_text_from_markdown.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_get_video_transcript_from_gcs.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_library.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_markdown_loader_gzip.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_markitdown_html.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_notebook_loader.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_pain_text.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_pdf_conversion_error.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_python_version_metadata.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_split_audio_with_llm.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_transcribe_s3_images_from_csv.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_transcribe_s3_images_from_csv_script.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_xml_xbrl_loader.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_youtube_gemini_minimal_check.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_youtube_llm_fallbacks.py +0 -0
- {polytext-0.2.8b2 → polytext-0.2.8b4}/tests/test_youtube_transcript.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: polytext
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.8b4
|
|
4
4
|
Summary: Python utilities to simplify document files management
|
|
5
5
|
Home-page: https://github.com/docsity/polytext
|
|
6
6
|
Author: Matteo Senardi
|
|
@@ -25,7 +25,7 @@ Requires-Dist: markdown-to-json==2.1.2
|
|
|
25
25
|
Requires-Dist: python-docx==1.1.2
|
|
26
26
|
Requires-Dist: google-api-core>=2.24.2
|
|
27
27
|
Requires-Dist: google-cloud-storage<3.0.0,>=2.17
|
|
28
|
-
Requires-Dist: google-genai
|
|
28
|
+
Requires-Dist: google-genai==2.22.0
|
|
29
29
|
Requires-Dist: openai==2.26.0
|
|
30
30
|
Requires-Dist: boto3>=1.42.64
|
|
31
31
|
Requires-Dist: botocore>=1.42.64
|
|
@@ -14,6 +14,7 @@ from google.genai import types
|
|
|
14
14
|
from google.genai import errors as genai_errors
|
|
15
15
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
16
16
|
from google.api_core import exceptions as google_exceptions
|
|
17
|
+
from pydub import AudioSegment
|
|
17
18
|
|
|
18
19
|
from ..exceptions import EmptyDocument
|
|
19
20
|
from ..prompts.transcription import (
|
|
@@ -25,13 +26,22 @@ from ..prompts.transcription import (
|
|
|
25
26
|
)
|
|
26
27
|
from ..processor.audio_chunker import AudioChunker
|
|
27
28
|
from ..processor.text_merger import TextMerger
|
|
28
|
-
from .gemini_quality_guards import
|
|
29
|
+
from .gemini_quality_guards import (
|
|
30
|
+
extract_finish_reason,
|
|
31
|
+
has_excessive_consecutive_word_repetition,
|
|
32
|
+
tail_has_excessive_repetition,
|
|
33
|
+
)
|
|
29
34
|
|
|
30
35
|
logger = logging.getLogger(__name__)
|
|
31
36
|
|
|
32
37
|
SUPPORTED_MIME_TYPES = {
|
|
33
38
|
'audio/x-aac', 'audio/flac', 'audio/mp3', 'audio/m4a', 'audio/mpeg',
|
|
34
|
-
'audio/mpga', 'audio/mp4', 'audio/opus', 'audio/pcm', 'audio/wav', 'audio/webm'
|
|
39
|
+
'audio/mpga', 'audio/mp4', 'audio/opus', 'audio/pcm', 'audio/wav', 'audio/x-wav', 'audio/webm'
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
GEMINI_AUDIO_MIME_ALIASES = {
|
|
43
|
+
'audio/x-aac': 'audio/aac',
|
|
44
|
+
'audio/x-wav': 'audio/wav',
|
|
35
45
|
}
|
|
36
46
|
|
|
37
47
|
INJECTION_GUARD_SYSTEM_INSTRUCTION = (
|
|
@@ -48,16 +58,20 @@ INJECTION_GUARD_SYSTEM_INSTRUCTION = (
|
|
|
48
58
|
)
|
|
49
59
|
|
|
50
60
|
AUDIO_MIN_OUTPUT_TOKENS = 500
|
|
61
|
+
AUDIO_DEFAULT_MAX_OUTPUT_TOKENS = 4096
|
|
62
|
+
AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH = 1
|
|
63
|
+
AUDIO_ADAPTIVE_SPLIT_OVERLAP_MS = 2000
|
|
64
|
+
AUDIO_LONG_DURATION_THRESHOLD_MS = 80 * 60 * 1000
|
|
51
65
|
AUDIO_TAIL_REPETITION_LINES = int(os.getenv("AUDIO_TAIL_REPETITION_LINES", "200"))
|
|
52
66
|
AUDIO_TAIL_REPETITION_THRESHOLD = float(os.getenv("AUDIO_TAIL_REPETITION_THRESHOLD", "0.35"))
|
|
53
67
|
AUDIO_FALLBACK_SOURCE_PATTERN = os.getenv("AUDIO_FALLBACK_SOURCE_PATTERN", "flash-lite")
|
|
54
|
-
AUDIO_FALLBACK_MODEL = os.getenv("AUDIO_FALLBACK_MODEL", "gemini-3-flash-
|
|
68
|
+
AUDIO_FALLBACK_MODEL = os.getenv("AUDIO_FALLBACK_MODEL", "gemini-3.5-flash-lite")
|
|
55
69
|
AUDIO_FALLBACK_TEMPERATURE = float(os.getenv("AUDIO_FALLBACK_TEMPERATURE", "1.0"))
|
|
56
70
|
AUDIO_FINAL_FALLBACK_MODEL = os.getenv("AUDIO_FINAL_FALLBACK_MODEL", "gemini-3.5-flash")
|
|
57
71
|
AUDIO_FILE_UPLOAD_THRESHOLD_BYTES = 20 * 1024 * 1024
|
|
58
72
|
AUDIO_PROMPT_VARIANT_DEFAULT = "default"
|
|
59
73
|
AUDIO_PROMPT_VARIANT_NON_LITERAL_FALLBACK = "non_literal_fallback"
|
|
60
|
-
AUDIO_RETRIABLE_OUTPUT_ERROR_CODES = (996, 997, 999)
|
|
74
|
+
AUDIO_RETRIABLE_OUTPUT_ERROR_CODES = (996, 997, 998, 999)
|
|
61
75
|
NO_HUMAN_SPEECH_MARKER = "no human speech detected"
|
|
62
76
|
|
|
63
77
|
|
|
@@ -80,23 +94,30 @@ def add_line_break_after_each_sentence(text: str) -> str:
|
|
|
80
94
|
if not text:
|
|
81
95
|
return text
|
|
82
96
|
|
|
97
|
+
line_feed = chr(10)
|
|
83
98
|
lines = text.splitlines()
|
|
84
99
|
formatted_lines = []
|
|
85
100
|
|
|
86
101
|
for line in lines:
|
|
87
102
|
stripped_line = line.strip()
|
|
103
|
+
|
|
88
104
|
if not stripped_line:
|
|
89
105
|
formatted_lines.append("")
|
|
90
106
|
continue
|
|
107
|
+
|
|
91
108
|
if re.match(r"^#{1,6}\s+", stripped_line):
|
|
92
109
|
formatted_lines.append(stripped_line)
|
|
93
110
|
continue
|
|
94
111
|
|
|
95
112
|
normalized_line = re.sub(r"\s+", " ", stripped_line)
|
|
96
|
-
normalized_line = re.sub(
|
|
113
|
+
normalized_line = re.sub(
|
|
114
|
+
r"([.!?])\s+",
|
|
115
|
+
lambda match: f"{match.group(1)}{line_feed} ",
|
|
116
|
+
normalized_line,
|
|
117
|
+
)
|
|
97
118
|
formatted_lines.append(normalized_line)
|
|
98
119
|
|
|
99
|
-
return
|
|
120
|
+
return line_feed.join(formatted_lines).strip()
|
|
100
121
|
|
|
101
122
|
|
|
102
123
|
def create_ascii_safe_upload_copy(audio_file: str) -> tuple[str, str | None]:
|
|
@@ -116,33 +137,32 @@ def create_ascii_safe_upload_copy(audio_file: str) -> tuple[str, str | None]:
|
|
|
116
137
|
|
|
117
138
|
def compress_and_convert_audio(input_path: str, bitrate_quality: int = 9) -> str:
|
|
118
139
|
"""
|
|
119
|
-
|
|
140
|
+
Normalize an audio file to lossless 16 kHz mono WAV using ffmpeg.
|
|
120
141
|
|
|
121
142
|
Args:
|
|
122
143
|
input_path (str): Path to the original audio file
|
|
123
|
-
bitrate_quality (int, optional):
|
|
144
|
+
bitrate_quality (int, optional): Retained for backward compatibility; WAV output is lossless.
|
|
124
145
|
|
|
125
146
|
Returns:
|
|
126
|
-
str: Path to the temporary
|
|
147
|
+
str: Path to the temporary normalized WAV file
|
|
127
148
|
|
|
128
149
|
Raises:
|
|
129
150
|
RuntimeError: If FFmpeg compression/conversion fails
|
|
130
151
|
|
|
131
152
|
Notes:
|
|
132
|
-
- Creates a temporary
|
|
133
|
-
- Converts audio to mono
|
|
153
|
+
- Creates a temporary WAV file that should be deleted after use
|
|
154
|
+
- Converts audio to 16-bit PCM mono at 16kHz
|
|
134
155
|
- Uses maximum available CPU threads for faster processing
|
|
135
156
|
"""
|
|
136
157
|
# Create temporary file for audio output
|
|
137
|
-
fd, temp_audio_path = tempfile.mkstemp(suffix='.
|
|
158
|
+
fd, temp_audio_path = tempfile.mkstemp(suffix='.wav')
|
|
138
159
|
os.close(fd)
|
|
139
160
|
|
|
140
|
-
logger.info(
|
|
161
|
+
logger.info("Normalizing audio to lossless 16 kHz mono WAV")
|
|
141
162
|
try:
|
|
142
163
|
ffmpeg.input(input_path).output(
|
|
143
164
|
temp_audio_path,
|
|
144
|
-
|
|
145
|
-
acodec='libmp3lame',
|
|
165
|
+
acodec='pcm_s16le',
|
|
146
166
|
ac=1, # Convert to mono
|
|
147
167
|
ar=16000, # Lower sample rate
|
|
148
168
|
vn=None,
|
|
@@ -155,7 +175,7 @@ def compress_and_convert_audio(input_path: str, bitrate_quality: int = 9) -> str
|
|
|
155
175
|
os.unlink(temp_audio_path)
|
|
156
176
|
raise
|
|
157
177
|
|
|
158
|
-
logger.info(f"Successfully
|
|
178
|
+
logger.info(f"Successfully normalized audio: {temp_audio_path}")
|
|
159
179
|
return temp_audio_path
|
|
160
180
|
|
|
161
181
|
|
|
@@ -182,7 +202,7 @@ def transcribe_full_audio(audio_file, markdown_output: bool = False,
|
|
|
182
202
|
bitrate_quality (int, optional): Variable bitrate quality from 0-9 (9 being lowest). Defaults to 9
|
|
183
203
|
timeout_minutes (int, optional): Number of minutes to wait for a response. Defaults to None.
|
|
184
204
|
max_llm_tokens (int, optional): Token budget used for audio chunk sizing. Defaults to 4250.
|
|
185
|
-
max_output_tokens (int | None, optional): Maximum Gemini output tokens. Defaults to
|
|
205
|
+
max_output_tokens (int | None, optional): Maximum Gemini output tokens. Defaults to 4096.
|
|
186
206
|
is_output_audio_raw (bool, optional): If True, use the raw Markdown audio prompt.
|
|
187
207
|
If False, use the formatted Markdown audio prompt. Defaults to True.
|
|
188
208
|
|
|
@@ -204,6 +224,8 @@ class AudioToTextConverter:
|
|
|
204
224
|
max_output_tokens: int | None = None, temp_dir: str = "temp",
|
|
205
225
|
bitrate_quality: int = 9, timeout_minutes: int = None,
|
|
206
226
|
fallback_stage: int = 0,
|
|
227
|
+
adaptive_split_depth: int = 0,
|
|
228
|
+
long_audio_protections_enabled: bool = False,
|
|
207
229
|
prompt_variant: str = AUDIO_PROMPT_VARIANT_DEFAULT,
|
|
208
230
|
is_output_audio_raw: bool = True):
|
|
209
231
|
"""
|
|
@@ -218,12 +240,14 @@ class AudioToTextConverter:
|
|
|
218
240
|
llm_api_key (str, optional): Override API key for language model. Defaults to None.
|
|
219
241
|
max_llm_tokens (int): Token budget used to size audio chunks. Defaults to 4250.
|
|
220
242
|
max_output_tokens (int | None): Maximum number of output tokens for Gemini generation.
|
|
221
|
-
Defaults to
|
|
243
|
+
Defaults to 4096.
|
|
222
244
|
temp_dir (str): Directory for temporary files. Defaults to "temp".
|
|
223
245
|
bitrate_quality (int, optional): Variable bitrate quality from 0-9 (9 being lowest). Defaults to 9
|
|
224
246
|
timeout_minutes (int): Number of minutes to wait for a response.
|
|
225
247
|
fallback_stage (int, optional): Internal retry stage used by fallback attempts.
|
|
226
248
|
Defaults to 0.
|
|
249
|
+
long_audio_protections_enabled (bool, optional): Apply stricter recovery rules inherited
|
|
250
|
+
from an original audio longer than 80 minutes. Defaults to False.
|
|
227
251
|
prompt_variant (str, optional): Prompt variant used by this attempt.
|
|
228
252
|
Defaults to "default".
|
|
229
253
|
is_output_audio_raw (bool, optional): If True, use the raw Markdown audio prompt.
|
|
@@ -240,12 +264,18 @@ class AudioToTextConverter:
|
|
|
240
264
|
self.markdown_output = markdown_output
|
|
241
265
|
self.llm_api_key = llm_api_key
|
|
242
266
|
self.max_llm_tokens = max(max_llm_tokens, AUDIO_MIN_OUTPUT_TOKENS)
|
|
243
|
-
requested_output_tokens =
|
|
267
|
+
requested_output_tokens = (
|
|
268
|
+
AUDIO_DEFAULT_MAX_OUTPUT_TOKENS
|
|
269
|
+
if max_output_tokens is None
|
|
270
|
+
else max_output_tokens
|
|
271
|
+
)
|
|
244
272
|
self.max_output_tokens = max(requested_output_tokens, AUDIO_MIN_OUTPUT_TOKENS)
|
|
245
273
|
self.chunked_audio = False
|
|
246
274
|
self.bitrate_quality = bitrate_quality
|
|
247
275
|
self.timeout_minutes = timeout_minutes
|
|
248
276
|
self.fallback_stage = fallback_stage
|
|
277
|
+
self.adaptive_split_depth = adaptive_split_depth
|
|
278
|
+
self.long_audio_protections_enabled = long_audio_protections_enabled
|
|
249
279
|
self.prompt_variant = prompt_variant
|
|
250
280
|
self.fallback_source_pattern = AUDIO_FALLBACK_SOURCE_PATTERN
|
|
251
281
|
self.fallback_model = AUDIO_FALLBACK_MODEL
|
|
@@ -269,6 +299,9 @@ class AudioToTextConverter:
|
|
|
269
299
|
return AUDIO_TO_MARKDOWN_PROMPT
|
|
270
300
|
return AUDIO_TO_PLAIN_TEXT_PROMPT
|
|
271
301
|
|
|
302
|
+
def set_long_audio_protections(self, duration_ms: int) -> None:
|
|
303
|
+
self.long_audio_protections_enabled = duration_ms > AUDIO_LONG_DURATION_THRESHOLD_MS
|
|
304
|
+
|
|
272
305
|
def should_prompt_fallback_retry(self, error: EmptyDocument) -> bool:
|
|
273
306
|
if self.fallback_stage != 0:
|
|
274
307
|
return False
|
|
@@ -331,6 +364,8 @@ class AudioToTextConverter:
|
|
|
331
364
|
bitrate_quality=self.bitrate_quality,
|
|
332
365
|
timeout_minutes=self.timeout_minutes,
|
|
333
366
|
fallback_stage=fallback_stage,
|
|
367
|
+
adaptive_split_depth=self.adaptive_split_depth,
|
|
368
|
+
long_audio_protections_enabled=self.long_audio_protections_enabled,
|
|
334
369
|
prompt_variant=resolved_prompt_variant,
|
|
335
370
|
is_output_audio_raw=self.is_output_audio_raw,
|
|
336
371
|
)
|
|
@@ -346,10 +381,83 @@ class AudioToTextConverter:
|
|
|
346
381
|
result.setdefault("fallback_to_prompt_variant", resolved_prompt_variant)
|
|
347
382
|
return result
|
|
348
383
|
|
|
384
|
+
def transcribe_audio_halves(self, audio_file: str, temperature: float = 0.0) -> dict:
|
|
385
|
+
"""Split one genuinely overlong chunk and transcribe both halves once."""
|
|
386
|
+
audio = AudioSegment.from_file(audio_file)
|
|
387
|
+
midpoint = len(audio) // 2
|
|
388
|
+
ranges = (
|
|
389
|
+
(0, min(len(audio), midpoint + AUDIO_ADAPTIVE_SPLIT_OVERLAP_MS)),
|
|
390
|
+
(max(0, midpoint - AUDIO_ADAPTIVE_SPLIT_OVERLAP_MS), len(audio)),
|
|
391
|
+
)
|
|
392
|
+
split_paths = []
|
|
393
|
+
split_results = []
|
|
394
|
+
|
|
395
|
+
try:
|
|
396
|
+
for start_ms, end_ms in ranges:
|
|
397
|
+
fd, split_path = tempfile.mkstemp(
|
|
398
|
+
prefix="adaptive-audio-split-",
|
|
399
|
+
suffix=".wav",
|
|
400
|
+
dir=self.temp_dir,
|
|
401
|
+
)
|
|
402
|
+
os.close(fd)
|
|
403
|
+
split_paths.append(split_path)
|
|
404
|
+
(
|
|
405
|
+
audio[start_ms:end_ms]
|
|
406
|
+
.set_frame_rate(16000)
|
|
407
|
+
.set_channels(1)
|
|
408
|
+
.set_sample_width(2)
|
|
409
|
+
.export(split_path, format="wav", codec="pcm_s16le")
|
|
410
|
+
.close()
|
|
411
|
+
)
|
|
412
|
+
|
|
413
|
+
split_converter = AudioToTextConverter(
|
|
414
|
+
transcription_model=self.transcription_model,
|
|
415
|
+
transcription_model_provider=self.transcription_model_provider,
|
|
416
|
+
k=self.k,
|
|
417
|
+
min_matches=self.min_matches,
|
|
418
|
+
markdown_output=self.markdown_output,
|
|
419
|
+
llm_api_key=self.llm_api_key,
|
|
420
|
+
max_llm_tokens=self.max_llm_tokens,
|
|
421
|
+
max_output_tokens=self.max_output_tokens,
|
|
422
|
+
temp_dir=self.temp_dir,
|
|
423
|
+
bitrate_quality=self.bitrate_quality,
|
|
424
|
+
timeout_minutes=self.timeout_minutes,
|
|
425
|
+
fallback_stage=self.fallback_stage,
|
|
426
|
+
adaptive_split_depth=self.adaptive_split_depth + 1,
|
|
427
|
+
long_audio_protections_enabled=self.long_audio_protections_enabled,
|
|
428
|
+
prompt_variant=self.prompt_variant,
|
|
429
|
+
is_output_audio_raw=self.is_output_audio_raw,
|
|
430
|
+
)
|
|
431
|
+
split_results.append(
|
|
432
|
+
split_converter.transcribe_audio(split_path, temperature=temperature)
|
|
433
|
+
)
|
|
434
|
+
|
|
435
|
+
merged_transcript = TextMerger(k=self.k, min_matches=self.min_matches).merge_texts(
|
|
436
|
+
split_results[0]["transcript"],
|
|
437
|
+
split_results[1]["transcript"],
|
|
438
|
+
)
|
|
439
|
+
return {
|
|
440
|
+
"transcript": merged_transcript,
|
|
441
|
+
"completion_tokens": sum(item["completion_tokens"] for item in split_results),
|
|
442
|
+
"prompt_tokens": sum(item["prompt_tokens"] for item in split_results),
|
|
443
|
+
"completion_model": self.transcription_model,
|
|
444
|
+
"completion_model_provider": self.transcription_model_provider,
|
|
445
|
+
"finish_reason": "ADAPTIVE_SPLIT",
|
|
446
|
+
"max_output_tokens": self.max_output_tokens,
|
|
447
|
+
"temperature": temperature,
|
|
448
|
+
"prompt_variant": self.prompt_variant,
|
|
449
|
+
"adaptive_split": True,
|
|
450
|
+
"split_results": split_results,
|
|
451
|
+
}
|
|
452
|
+
finally:
|
|
453
|
+
for split_path in split_paths:
|
|
454
|
+
if os.path.exists(split_path):
|
|
455
|
+
os.remove(split_path)
|
|
456
|
+
|
|
349
457
|
def build_config(self, output_budget: int, temperature: float = 0.0) -> types.GenerateContentConfig:
|
|
350
458
|
return types.GenerateContentConfig(
|
|
351
459
|
temperature=temperature,
|
|
352
|
-
thinking_config=types.ThinkingConfig(
|
|
460
|
+
thinking_config=types.ThinkingConfig(thinking_level="minimal"),
|
|
353
461
|
max_output_tokens=output_budget,
|
|
354
462
|
system_instruction=INJECTION_GUARD_SYSTEM_INSTRUCTION,
|
|
355
463
|
tools=[],
|
|
@@ -426,6 +534,7 @@ class AudioToTextConverter:
|
|
|
426
534
|
except ValueError:
|
|
427
535
|
logger.exception("Unsupported audio format for %s", audio_file)
|
|
428
536
|
raise
|
|
537
|
+
mime_type = GEMINI_AUDIO_MIME_ALIASES.get(mime_type, mime_type)
|
|
429
538
|
|
|
430
539
|
return client.models.generate_content(
|
|
431
540
|
model=self.transcription_model,
|
|
@@ -448,7 +557,6 @@ class AudioToTextConverter:
|
|
|
448
557
|
google_exceptions.ServiceUnavailable,
|
|
449
558
|
google_exceptions.InternalServerError,
|
|
450
559
|
genai_errors.ServerError,
|
|
451
|
-
genai_errors.APIError,
|
|
452
560
|
),
|
|
453
561
|
tries=8,
|
|
454
562
|
delay=1,
|
|
@@ -511,6 +619,9 @@ class AudioToTextConverter:
|
|
|
511
619
|
tail_lines=AUDIO_TAIL_REPETITION_LINES,
|
|
512
620
|
threshold=AUDIO_TAIL_REPETITION_THRESHOLD,
|
|
513
621
|
)
|
|
622
|
+
has_repetitive_word_loop = has_excessive_consecutive_word_repetition(
|
|
623
|
+
response_text,
|
|
624
|
+
)
|
|
514
625
|
usage_metadata = getattr(response, "usage_metadata", None)
|
|
515
626
|
completion_tokens = getattr(usage_metadata, "candidates_token_count", 0) or 0
|
|
516
627
|
prompt_tokens = getattr(usage_metadata, "prompt_token_count", 0) or 0
|
|
@@ -525,20 +636,83 @@ class AudioToTextConverter:
|
|
|
525
636
|
)
|
|
526
637
|
|
|
527
638
|
if finish_reason and "MAX_TOKENS" in finish_reason:
|
|
639
|
+
if (
|
|
640
|
+
self.long_audio_protections_enabled
|
|
641
|
+
and self.adaptive_split_depth < AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH
|
|
642
|
+
):
|
|
643
|
+
logger.info(
|
|
644
|
+
"Splitting long-audio chunk before fallback after MAX_TOKENS response: %s",
|
|
645
|
+
audio_file,
|
|
646
|
+
)
|
|
647
|
+
return self.transcribe_audio_halves(audio_file, temperature=temperature)
|
|
648
|
+
if has_repetitive_tail or has_repetitive_word_loop:
|
|
649
|
+
raise EmptyDocument(
|
|
650
|
+
message=f"Transcript discarded because repetitive output reached max tokens for audio: {audio_file}",
|
|
651
|
+
code=997,
|
|
652
|
+
)
|
|
653
|
+
if self.adaptive_split_depth < AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH:
|
|
654
|
+
logger.info(
|
|
655
|
+
"Splitting audio chunk after non-repetitive MAX_TOKENS response: %s",
|
|
656
|
+
audio_file,
|
|
657
|
+
)
|
|
658
|
+
return self.transcribe_audio_halves(audio_file, temperature=temperature)
|
|
528
659
|
raise EmptyDocument(
|
|
529
660
|
message=f"Transcript truncated because max output tokens were reached for audio: {audio_file}",
|
|
530
661
|
code=999,
|
|
531
662
|
)
|
|
532
663
|
|
|
533
664
|
if has_repetitive_tail:
|
|
665
|
+
if (
|
|
666
|
+
self.long_audio_protections_enabled
|
|
667
|
+
and self.adaptive_split_depth < AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH
|
|
668
|
+
):
|
|
669
|
+
logger.info(
|
|
670
|
+
"Splitting long-audio chunk before fallback after repetitive tail: %s",
|
|
671
|
+
audio_file,
|
|
672
|
+
)
|
|
673
|
+
return self.transcribe_audio_halves(audio_file, temperature=temperature)
|
|
534
674
|
raise EmptyDocument(
|
|
535
675
|
message=f"Transcript discarded because repetitive tail was detected for audio: {audio_file}",
|
|
536
676
|
code=997,
|
|
537
677
|
)
|
|
538
678
|
|
|
679
|
+
if has_repetitive_word_loop:
|
|
680
|
+
if (
|
|
681
|
+
self.long_audio_protections_enabled
|
|
682
|
+
and self.adaptive_split_depth < AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH
|
|
683
|
+
):
|
|
684
|
+
logger.info(
|
|
685
|
+
"Splitting long-audio chunk before fallback after repetitive word loop: %s",
|
|
686
|
+
audio_file,
|
|
687
|
+
)
|
|
688
|
+
return self.transcribe_audio_halves(audio_file, temperature=temperature)
|
|
689
|
+
raise EmptyDocument(
|
|
690
|
+
message=f"Transcript discarded because repetitive word loop was detected for audio: {audio_file}",
|
|
691
|
+
code=997,
|
|
692
|
+
)
|
|
693
|
+
|
|
539
694
|
response_text, marker_only = normalize_no_human_speech_marker(response_text)
|
|
540
|
-
|
|
541
|
-
|
|
695
|
+
|
|
696
|
+
is_insignificant_stop = (
|
|
697
|
+
finish_reason
|
|
698
|
+
and "STOP" in finish_reason
|
|
699
|
+
and (
|
|
700
|
+
not response_text.strip()
|
|
701
|
+
or (
|
|
702
|
+
completion_tokens <= 4
|
|
703
|
+
and len(response_text.split()) <= 2
|
|
704
|
+
)
|
|
705
|
+
)
|
|
706
|
+
)
|
|
707
|
+
if (
|
|
708
|
+
self.long_audio_protections_enabled
|
|
709
|
+
and not marker_only
|
|
710
|
+
and is_insignificant_stop
|
|
711
|
+
):
|
|
712
|
+
raise EmptyDocument(
|
|
713
|
+
message=f"Transcript discarded because STOP returned empty or insignificant output for audio: {audio_file}",
|
|
714
|
+
code=998,
|
|
715
|
+
)
|
|
542
716
|
|
|
543
717
|
response_dict = {
|
|
544
718
|
"transcript": "" if marker_only else response_text,
|
|
@@ -557,6 +731,24 @@ class AudioToTextConverter:
|
|
|
557
731
|
)
|
|
558
732
|
return response_dict
|
|
559
733
|
except EmptyDocument as e:
|
|
734
|
+
if (
|
|
735
|
+
e.code == 999
|
|
736
|
+
and self.adaptive_split_depth >= AUDIO_MAX_ADAPTIVE_SPLIT_DEPTH
|
|
737
|
+
and self.fallback_stage == 0
|
|
738
|
+
and self.transcription_model != self.fallback_model
|
|
739
|
+
):
|
|
740
|
+
return self.run_fallback(
|
|
741
|
+
audio_file=audio_file,
|
|
742
|
+
reason=e.message,
|
|
743
|
+
fallback_model=self.fallback_model,
|
|
744
|
+
fallback_temperature=self.fallback_temperature,
|
|
745
|
+
fallback_stage=2 if self.markdown_output else 1,
|
|
746
|
+
prompt_variant=(
|
|
747
|
+
AUDIO_PROMPT_VARIANT_NON_LITERAL_FALLBACK
|
|
748
|
+
if self.markdown_output
|
|
749
|
+
else self.prompt_variant
|
|
750
|
+
),
|
|
751
|
+
)
|
|
560
752
|
if self.should_prompt_fallback_retry(e):
|
|
561
753
|
return self.run_fallback(
|
|
562
754
|
audio_file=audio_file,
|
|
@@ -649,6 +841,13 @@ class AudioToTextConverter:
|
|
|
649
841
|
# Create chunker and extract chunks
|
|
650
842
|
logger.info("Creating AudioChunker instance...")
|
|
651
843
|
chunker = AudioChunker(used_file, max_llm_tokens=self.max_llm_tokens)
|
|
844
|
+
self.set_long_audio_protections(chunker.duration_ms)
|
|
845
|
+
logger.info(
|
|
846
|
+
"Long-audio transcription protections enabled: %s (duration: %sms, threshold: %sms)",
|
|
847
|
+
self.long_audio_protections_enabled,
|
|
848
|
+
chunker.duration_ms,
|
|
849
|
+
AUDIO_LONG_DURATION_THRESHOLD_MS,
|
|
850
|
+
)
|
|
652
851
|
chunks = chunker.extract_chunks()
|
|
653
852
|
|
|
654
853
|
logger.info(f"chunks: {chunks}")
|
|
@@ -9,8 +9,6 @@ from google.api_core import exceptions as google_exceptions
|
|
|
9
9
|
from retry import retry
|
|
10
10
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
11
11
|
|
|
12
|
-
from polytext.processor.transcript_chunker import TranscriptChunker
|
|
13
|
-
from polytext.processor.text_merger import TextMerger
|
|
14
12
|
from polytext.prompts.beautiful_text import BEAUTIFUL_TEXT_PROMPT
|
|
15
13
|
|
|
16
14
|
logger = logging.getLogger(__name__)
|
|
@@ -26,6 +24,7 @@ class BeautifulTextConverter:
|
|
|
26
24
|
prompt_overhead: int = 1800,
|
|
27
25
|
tokens_per_char: float = 0.25,
|
|
28
26
|
overlap_chars: int = 800,
|
|
27
|
+
max_target_chars: int = 12000,
|
|
29
28
|
) -> None:
|
|
30
29
|
self.llm_api_key = llm_api_key
|
|
31
30
|
self.model = model
|
|
@@ -34,19 +33,58 @@ class BeautifulTextConverter:
|
|
|
34
33
|
self.prompt_overhead = prompt_overhead
|
|
35
34
|
self.tokens_per_char = tokens_per_char
|
|
36
35
|
self.overlap_chars = overlap_chars
|
|
36
|
+
self.max_target_chars = max_target_chars
|
|
37
37
|
|
|
38
38
|
def get_client(self):
|
|
39
39
|
return genai.Client(api_key=self.llm_api_key) if self.llm_api_key else genai.Client()
|
|
40
40
|
|
|
41
41
|
def chunk_raw_text(self, raw_text: str) -> list[dict]:
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
42
|
+
text = (raw_text or "").strip()
|
|
43
|
+
if not text:
|
|
44
|
+
return []
|
|
45
|
+
|
|
46
|
+
chunks = []
|
|
47
|
+
start = 0
|
|
48
|
+
index = 0
|
|
49
|
+
|
|
50
|
+
while start < len(text):
|
|
51
|
+
proposed_end = min(start + self.max_target_chars, len(text))
|
|
52
|
+
end = self._find_chunk_end(text, start, proposed_end)
|
|
53
|
+
target_text = text[start:end].strip()
|
|
54
|
+
|
|
55
|
+
if not target_text:
|
|
56
|
+
break
|
|
57
|
+
|
|
58
|
+
chunks.append(
|
|
59
|
+
{
|
|
60
|
+
"index": index,
|
|
61
|
+
"target_text": target_text,
|
|
62
|
+
}
|
|
63
|
+
)
|
|
64
|
+
start = end
|
|
65
|
+
index += 1
|
|
66
|
+
|
|
67
|
+
return chunks
|
|
68
|
+
|
|
69
|
+
def _find_chunk_end(self, text: str, start: int, proposed_end: int) -> int:
|
|
70
|
+
if proposed_end >= len(text):
|
|
71
|
+
return len(text)
|
|
72
|
+
|
|
73
|
+
minimum_end = start + int(self.max_target_chars * 0.65)
|
|
74
|
+
chunk_window = text[minimum_end:proposed_end]
|
|
75
|
+
sentence_boundaries = list(re.finditer(r"(?<=[.!?])\s+", chunk_window))
|
|
76
|
+
if sentence_boundaries:
|
|
77
|
+
return minimum_end + sentence_boundaries[-1].end()
|
|
78
|
+
|
|
79
|
+
paragraph_boundary = text.rfind("\n\n", minimum_end, proposed_end)
|
|
80
|
+
if paragraph_boundary != -1:
|
|
81
|
+
return paragraph_boundary + 2
|
|
82
|
+
|
|
83
|
+
whitespace_boundary = text.rfind(" ", minimum_end, proposed_end)
|
|
84
|
+
if whitespace_boundary != -1:
|
|
85
|
+
return whitespace_boundary + 1
|
|
86
|
+
|
|
87
|
+
return proposed_end
|
|
50
88
|
|
|
51
89
|
@retry(
|
|
52
90
|
(
|
|
@@ -60,11 +98,18 @@ class BeautifulTextConverter:
|
|
|
60
98
|
backoff=2,
|
|
61
99
|
logger=logger,
|
|
62
100
|
)
|
|
63
|
-
def process_chunk(self, client,
|
|
101
|
+
def process_chunk(self, client, chunk: dict, index: int) -> dict:
|
|
64
102
|
logger.info("Processing beautiful text chunk %s", index + 1)
|
|
65
103
|
start_time = time.time()
|
|
66
104
|
|
|
105
|
+
target_text = chunk["target_text"]
|
|
106
|
+
|
|
67
107
|
config = types.GenerateContentConfig(
|
|
108
|
+
temperature=0,
|
|
109
|
+
thinking_config=types.ThinkingConfig(thinking_budget=0),
|
|
110
|
+
max_output_tokens=self.max_llm_tokens,
|
|
111
|
+
tools=[],
|
|
112
|
+
automatic_function_calling=types.AutomaticFunctionCallingConfig(disable=True),
|
|
68
113
|
safety_settings=[
|
|
69
114
|
types.SafetySetting(
|
|
70
115
|
category=types.HarmCategory.HARM_CATEGORY_HATE_SPEECH,
|
|
@@ -87,7 +132,11 @@ class BeautifulTextConverter:
|
|
|
87
132
|
|
|
88
133
|
response = client.models.generate_content(
|
|
89
134
|
model=self.model,
|
|
90
|
-
contents=[
|
|
135
|
+
contents=[
|
|
136
|
+
BEAUTIFUL_TEXT_PROMPT,
|
|
137
|
+
"TARGET TEXT TO CLEAN",
|
|
138
|
+
target_text,
|
|
139
|
+
],
|
|
91
140
|
config=config,
|
|
92
141
|
)
|
|
93
142
|
|
|
@@ -100,7 +149,7 @@ class BeautifulTextConverter:
|
|
|
100
149
|
}
|
|
101
150
|
|
|
102
151
|
def merge_cleaned_chunks(self, chunks: list[str]) -> str:
|
|
103
|
-
return
|
|
152
|
+
return "\n\n".join(chunk.strip() for chunk in chunks if chunk.strip())
|
|
104
153
|
|
|
105
154
|
def _convert_markdown_to_json(self, markdown_text: str) -> dict:
|
|
106
155
|
if not markdown_text.strip():
|
|
@@ -181,7 +230,7 @@ class BeautifulTextConverter:
|
|
|
181
230
|
|
|
182
231
|
with ThreadPoolExecutor() as executor:
|
|
183
232
|
future_to_index = {
|
|
184
|
-
executor.submit(self.process_chunk, client, chunk
|
|
233
|
+
executor.submit(self.process_chunk, client, chunk, chunk["index"]): chunk["index"]
|
|
185
234
|
for chunk in chunks
|
|
186
235
|
}
|
|
187
236
|
|
|
@@ -242,7 +242,9 @@ class DocumentOCRToTextConverter:
|
|
|
242
242
|
return error.code in OCR_RETRIABLE_OUTPUT_ERROR_CODES
|
|
243
243
|
|
|
244
244
|
def should_fallback_temperature_retry(self, error: EmptyDocument, temperature: float) -> bool:
|
|
245
|
-
|
|
245
|
+
# The non-literal prompt retry advances every OCR mode to stage 1.
|
|
246
|
+
# Plain-text document OCR must therefore also try the fallback model at stage 1.
|
|
247
|
+
expected_stage = 1
|
|
246
248
|
if self.fallback_stage != expected_stage:
|
|
247
249
|
return False
|
|
248
250
|
if error.code not in OCR_RETRIABLE_OUTPUT_ERROR_CODES:
|
|
@@ -368,6 +370,8 @@ class DocumentOCRToTextConverter:
|
|
|
368
370
|
config = types.GenerateContentConfig(
|
|
369
371
|
temperature=temperature,
|
|
370
372
|
max_output_tokens=self.max_output_tokens,
|
|
373
|
+
tools=[],
|
|
374
|
+
automatic_function_calling=types.AutomaticFunctionCallingConfig(disable=True),
|
|
371
375
|
safety_settings=[
|
|
372
376
|
types.SafetySetting(
|
|
373
377
|
category=types.HarmCategory.HARM_CATEGORY_HATE_SPEECH,
|
|
@@ -37,6 +37,14 @@ def has_consecutive_repetition(items: list[str], min_run_length: int = 3) -> boo
|
|
|
37
37
|
return False
|
|
38
38
|
|
|
39
39
|
|
|
40
|
+
def has_excessive_consecutive_word_repetition(
|
|
41
|
+
text: str,
|
|
42
|
+
min_run_length: int = 12,
|
|
43
|
+
) -> bool:
|
|
44
|
+
words = re.findall(r"\b\w+\b", (text or "").casefold())
|
|
45
|
+
return has_consecutive_repetition(words, min_run_length=min_run_length)
|
|
46
|
+
|
|
47
|
+
|
|
40
48
|
def tail_has_excessive_repetition(
|
|
41
49
|
text: str,
|
|
42
50
|
tail_lines: int,
|
|
@@ -352,6 +352,8 @@ class OCRToTextConverter:
|
|
|
352
352
|
config = types.GenerateContentConfig(
|
|
353
353
|
temperature=temperature,
|
|
354
354
|
max_output_tokens=self.max_output_tokens,
|
|
355
|
+
tools=[],
|
|
356
|
+
automatic_function_calling=types.AutomaticFunctionCallingConfig(disable=True),
|
|
355
357
|
safety_settings=[
|
|
356
358
|
types.SafetySetting(
|
|
357
359
|
category=types.HarmCategory.HARM_CATEGORY_HATE_SPEECH,
|
|
@@ -141,6 +141,8 @@ class TextToMdConverter:
|
|
|
141
141
|
start_time = time.time()
|
|
142
142
|
|
|
143
143
|
config = types.GenerateContentConfig(
|
|
144
|
+
tools=[],
|
|
145
|
+
automatic_function_calling=types.AutomaticFunctionCallingConfig(disable=True),
|
|
144
146
|
safety_settings=[
|
|
145
147
|
types.SafetySetting(category=types.HarmCategory.HARM_CATEGORY_HATE_SPEECH, threshold=types.HarmBlockThreshold.BLOCK_NONE),
|
|
146
148
|
types.SafetySetting(category=types.HarmCategory.HARM_CATEGORY_DANGEROUS_CONTENT, threshold=types.HarmBlockThreshold.BLOCK_NONE),
|
|
@@ -235,4 +237,4 @@ class TextToMdConverter:
|
|
|
235
237
|
|
|
236
238
|
logging.info(f"**** FINISH ****")
|
|
237
239
|
|
|
238
|
-
return result_dict
|
|
240
|
+
return result_dict
|