bilinote-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bilinote_cli-0.1.0.dist-info/METADATA +205 -0
- bilinote_cli-0.1.0.dist-info/RECORD +89 -0
- bilinote_cli-0.1.0.dist-info/WHEEL +4 -0
- bilinote_cli-0.1.0.dist-info/entry_points.txt +2 -0
- bilinote_cli-0.1.0.dist-info/licenses/LICENSE +21 -0
- src/__init__.py +0 -0
- src/app/config_manager.py +287 -0
- src/app/core/__init__.py +0 -0
- src/app/decorators/__init__.py +0 -0
- src/app/decorators/timeit.py +16 -0
- src/app/downloaders/__init__.py +0 -0
- src/app/downloaders/base.py +52 -0
- src/app/downloaders/bilibili_downloader.py +365 -0
- src/app/downloaders/common.py +1 -0
- src/app/downloaders/douyin_downloader.py +294 -0
- src/app/downloaders/douyin_helper/abogus.py +635 -0
- src/app/downloaders/kuaishou_downloader.py +93 -0
- src/app/downloaders/kuaishou_helper/__init__.py +0 -0
- src/app/downloaders/kuaishou_helper/kuaishou.py +96 -0
- src/app/downloaders/local_downloader.py +137 -0
- src/app/downloaders/xiaoyuzhoufm_download.py +25 -0
- src/app/downloaders/youtube_downloader.py +121 -0
- src/app/downloaders/youtube_subtitle.py +98 -0
- src/app/enmus/exception.py +21 -0
- src/app/enmus/note_enums.py +7 -0
- src/app/enmus/task_status_enums.py +28 -0
- src/app/exceptions/__init__.py +0 -0
- src/app/exceptions/biz_exception.py +6 -0
- src/app/exceptions/note.py +9 -0
- src/app/exceptions/provider.py +12 -0
- src/app/gpt/__init__.py +0 -0
- src/app/gpt/base.py +17 -0
- src/app/gpt/gpt_factory.py +13 -0
- src/app/gpt/prompt.py +77 -0
- src/app/gpt/prompt_builder.py +114 -0
- src/app/gpt/provider/OpenAI_compatible_provider.py +31 -0
- src/app/gpt/request_chunker.py +161 -0
- src/app/gpt/universal_gpt.py +348 -0
- src/app/gpt/utils.py +4 -0
- src/app/models/__init__.py +0 -0
- src/app/models/audio_model.py +15 -0
- src/app/models/gpt_model.py +19 -0
- src/app/models/model_config.py +16 -0
- src/app/models/notes_model.py +12 -0
- src/app/models/pipeline_model.py +27 -0
- src/app/models/provide_model.py +16 -0
- src/app/models/transcriber_model.py +16 -0
- src/app/models/video_record.py +0 -0
- src/app/secret_manager.py +116 -0
- src/app/services/__init__.py +0 -0
- src/app/services/batch_processor.py +343 -0
- src/app/services/cache/__init__.py +0 -0
- src/app/services/cache/task_cache.py +122 -0
- src/app/services/constant.py +17 -0
- src/app/services/note.py +141 -0
- src/app/services/pipeline/__init__.py +0 -0
- src/app/services/pipeline/ai_processor.py +98 -0
- src/app/services/pipeline/preparer.py +264 -0
- src/app/services/postprocessing.py +54 -0
- src/app/services/searcher.py +113 -0
- src/app/services/serial_executor.py +17 -0
- src/app/transcriber/__init__.py +0 -0
- src/app/transcriber/base.py +14 -0
- src/app/transcriber/bcut.py +245 -0
- src/app/transcriber/groq.py +76 -0
- src/app/transcriber/kuaishou.py +109 -0
- src/app/transcriber/transcriber_provider.py +134 -0
- src/app/transcriber/whisper_cpp.py +120 -0
- src/app/utils/cookie_helper.py +18 -0
- src/app/utils/env_checker.py +0 -0
- src/app/utils/file_cleanup.py +36 -0
- src/app/utils/logger.py +37 -0
- src/app/utils/note_helper.py +66 -0
- src/app/utils/path_helper.py +150 -0
- src/app/utils/screenshot_marker.py +13 -0
- src/app/utils/status_code.py +12 -0
- src/app/utils/url_parser.py +81 -0
- src/app/utils/video_helper.py +70 -0
- src/app/utils/video_reader.py +188 -0
- src/app/validators/__init__.py +0 -0
- src/app/validators/video_url_validator.py +38 -0
- src/cli.py +613 -0
- src/config/__init__.py +0 -0
- src/config/config.yaml.example +37 -0
- src/config/dev_config.json +6 -0
- src/config/model_config_manager.py +218 -0
- src/config/models.json +84 -0
- src/config/transcriber.json +29 -0
- src/ffmpeg_helper.py +49 -0
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
import enum
|
|
2
|
+
|
|
3
|
+
from abc import ABC, abstractmethod
|
|
4
|
+
from typing import Optional, Union
|
|
5
|
+
|
|
6
|
+
from app.enmus.note_enums import DownloadQuality
|
|
7
|
+
from app.models.notes_model import AudioDownloadResult
|
|
8
|
+
from app.models.transcriber_model import TranscriptResult
|
|
9
|
+
from os import getenv
|
|
10
|
+
QUALITY_MAP = {
|
|
11
|
+
"fast": "32",
|
|
12
|
+
"medium": "64",
|
|
13
|
+
"slow": "128"
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class Downloader(ABC):
|
|
18
|
+
def __init__(self):
|
|
19
|
+
#TODO 需要修改为可配置
|
|
20
|
+
self.quality = QUALITY_MAP.get('fast')
|
|
21
|
+
self.cache_data=getenv('DATA_DIR')
|
|
22
|
+
|
|
23
|
+
@abstractmethod
|
|
24
|
+
def download(self, video_url: str, output_dir: str = None,
|
|
25
|
+
quality: DownloadQuality = "fast", need_video: Optional[bool] = False,
|
|
26
|
+
skip_download: bool = False) -> AudioDownloadResult:
|
|
27
|
+
'''
|
|
28
|
+
|
|
29
|
+
:param need_video:
|
|
30
|
+
:param video_url: 资源链接
|
|
31
|
+
:param output_dir: 输出路径 默认根目录data
|
|
32
|
+
:param quality: 音频质量 fast | medium | slow
|
|
33
|
+
:return:返回一个 AudioDownloadResult 类
|
|
34
|
+
'''
|
|
35
|
+
pass
|
|
36
|
+
|
|
37
|
+
@staticmethod
|
|
38
|
+
def download_video(self, video_url: str,
|
|
39
|
+
output_dir: Union[str, None] = None) -> str:
|
|
40
|
+
pass
|
|
41
|
+
|
|
42
|
+
def download_subtitles(self, video_url: str, output_dir: str = None,
|
|
43
|
+
langs: list = None) -> Optional[TranscriptResult]:
|
|
44
|
+
'''
|
|
45
|
+
尝试获取平台字幕(人工字幕或自动生成字幕)
|
|
46
|
+
|
|
47
|
+
:param video_url: 视频链接
|
|
48
|
+
:param output_dir: 输出路径
|
|
49
|
+
:param langs: 优先语言列表,如 ['zh-Hans', 'zh', 'en']
|
|
50
|
+
:return: TranscriptResult 或 None(无字幕时)
|
|
51
|
+
'''
|
|
52
|
+
return None
|
|
@@ -0,0 +1,365 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import json
|
|
3
|
+
import logging
|
|
4
|
+
import tempfile
|
|
5
|
+
from abc import ABC
|
|
6
|
+
from typing import Union, Optional, List
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
import yt_dlp
|
|
10
|
+
|
|
11
|
+
from app.downloaders.base import Downloader, DownloadQuality, QUALITY_MAP
|
|
12
|
+
from app.models.notes_model import AudioDownloadResult
|
|
13
|
+
from app.models.transcriber_model import TranscriptResult, TranscriptSegment
|
|
14
|
+
from app.utils.path_helper import get_path_manager
|
|
15
|
+
from app.utils.url_parser import extract_video_id
|
|
16
|
+
from app.utils.cookie_helper import get_cookie
|
|
17
|
+
|
|
18
|
+
logger = logging.getLogger(__name__)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _cookie_string_to_file(cookie_str: str, domain: str = ".bilibili.com") -> str:
|
|
22
|
+
"""将 cookie 字符串转换为 Netscape 格式的临时文件,供 yt-dlp 使用"""
|
|
23
|
+
tmp_fd, tmp_path = tempfile.mkstemp(suffix=".txt", prefix="bili_cookie_")
|
|
24
|
+
# 使用一个足够远的未来时间戳作为 cookie 过期时间
|
|
25
|
+
expires = "2147483647"
|
|
26
|
+
with os.fdopen(tmp_fd, 'w', encoding='utf-8') as f:
|
|
27
|
+
f.write("# Netscape HTTP Cookie File\n")
|
|
28
|
+
for item in cookie_str.split(';'):
|
|
29
|
+
item = item.strip()
|
|
30
|
+
if '=' in item:
|
|
31
|
+
name, value = item.split('=', 1)
|
|
32
|
+
name = name.strip()
|
|
33
|
+
value = value.strip()
|
|
34
|
+
if name and value:
|
|
35
|
+
f.write(f"{domain}\tTRUE\t/\tFALSE\t{expires}\t{name}\t{value}\n")
|
|
36
|
+
return tmp_path
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _apply_bilibili_cookie(ydl_opts: dict):
|
|
40
|
+
"""统一为 yt-dlp 选项添加 B站 cookie 支持"""
|
|
41
|
+
cookie_str = get_cookie('bilibili')
|
|
42
|
+
if cookie_str:
|
|
43
|
+
cookie_file = _cookie_string_to_file(cookie_str)
|
|
44
|
+
ydl_opts['cookiefile'] = cookie_file
|
|
45
|
+
return cookie_file
|
|
46
|
+
return None
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class BilibiliDownloader(Downloader, ABC):
|
|
50
|
+
def __init__(self):
|
|
51
|
+
super().__init__()
|
|
52
|
+
|
|
53
|
+
def download(
|
|
54
|
+
self,
|
|
55
|
+
video_url: str,
|
|
56
|
+
output_dir: Union[str, None] = None,
|
|
57
|
+
quality: DownloadQuality = "fast",
|
|
58
|
+
need_video: Optional[bool] = False,
|
|
59
|
+
skip_download: bool = False,
|
|
60
|
+
) -> AudioDownloadResult:
|
|
61
|
+
if output_dir is None:
|
|
62
|
+
output_dir = get_path_manager().downloads_dir
|
|
63
|
+
os.makedirs(output_dir, exist_ok=True)
|
|
64
|
+
|
|
65
|
+
output_path = os.path.join(output_dir, "%(id)s.%(ext)s")
|
|
66
|
+
|
|
67
|
+
ydl_opts = {
|
|
68
|
+
'format': 'bestaudio[ext=m4a]/bestaudio/best',
|
|
69
|
+
'outtmpl': output_path,
|
|
70
|
+
'postprocessors': [
|
|
71
|
+
{
|
|
72
|
+
'key': 'FFmpegExtractAudio',
|
|
73
|
+
'preferredcodec': 'mp3',
|
|
74
|
+
'preferredquality': '64',
|
|
75
|
+
}
|
|
76
|
+
],
|
|
77
|
+
'noplaylist': True,
|
|
78
|
+
'quiet': False,
|
|
79
|
+
'http_headers': {
|
|
80
|
+
'Referer': 'https://www.bilibili.com',
|
|
81
|
+
'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36',
|
|
82
|
+
},
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
cookie_file = _apply_bilibili_cookie(ydl_opts)
|
|
86
|
+
|
|
87
|
+
if skip_download:
|
|
88
|
+
ydl_opts['skip_download'] = True
|
|
89
|
+
|
|
90
|
+
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
|
91
|
+
info = ydl.extract_info(video_url, download=not skip_download)
|
|
92
|
+
video_id = info.get("id")
|
|
93
|
+
title = info.get("title")
|
|
94
|
+
duration = info.get("duration", 0)
|
|
95
|
+
cover_url = info.get("thumbnail")
|
|
96
|
+
audio_path = os.path.join(output_dir, f"{video_id}.mp3")
|
|
97
|
+
|
|
98
|
+
# 清理临时 cookie 文件
|
|
99
|
+
if cookie_file and os.path.exists(cookie_file):
|
|
100
|
+
os.unlink(cookie_file)
|
|
101
|
+
|
|
102
|
+
if not skip_download and not os.path.exists(audio_path):
|
|
103
|
+
raise FileNotFoundError(f"音频下载失败,文件未生成: {audio_path}")
|
|
104
|
+
|
|
105
|
+
return AudioDownloadResult(
|
|
106
|
+
file_path=audio_path,
|
|
107
|
+
title=title,
|
|
108
|
+
duration=duration,
|
|
109
|
+
cover_url=cover_url,
|
|
110
|
+
platform="bilibili",
|
|
111
|
+
video_id=video_id,
|
|
112
|
+
raw_info=info,
|
|
113
|
+
video_path=None # ❗音频下载不包含视频路径
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
def download_video(
|
|
117
|
+
self,
|
|
118
|
+
video_url: str,
|
|
119
|
+
output_dir: Union[str, None] = None,
|
|
120
|
+
) -> str:
|
|
121
|
+
"""
|
|
122
|
+
下载视频,返回视频文件路径
|
|
123
|
+
"""
|
|
124
|
+
|
|
125
|
+
if output_dir is None:
|
|
126
|
+
output_dir = get_path_manager().downloads_dir
|
|
127
|
+
os.makedirs(output_dir, exist_ok=True)
|
|
128
|
+
print("video_url",video_url)
|
|
129
|
+
video_id=extract_video_id(video_url, "bilibili")
|
|
130
|
+
video_path = os.path.join(output_dir, f"{video_id}.mp4")
|
|
131
|
+
if os.path.exists(video_path):
|
|
132
|
+
return video_path
|
|
133
|
+
|
|
134
|
+
# 检查是否已经存在
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
output_path = os.path.join(output_dir, "%(id)s.%(ext)s")
|
|
138
|
+
|
|
139
|
+
ydl_opts = {
|
|
140
|
+
'format': 'bv*[ext=mp4]/bestvideo+bestaudio/best',
|
|
141
|
+
'outtmpl': output_path,
|
|
142
|
+
'noplaylist': True,
|
|
143
|
+
'quiet': False,
|
|
144
|
+
'merge_output_format': 'mp4', # 确保合并成 mp4
|
|
145
|
+
'http_headers': {
|
|
146
|
+
'Referer': 'https://www.bilibili.com',
|
|
147
|
+
'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36',
|
|
148
|
+
},
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
cookie_file = _apply_bilibili_cookie(ydl_opts)
|
|
152
|
+
|
|
153
|
+
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
|
154
|
+
info = ydl.extract_info(video_url, download=True)
|
|
155
|
+
video_id = info.get("id")
|
|
156
|
+
video_path = os.path.join(output_dir, f"{video_id}.mp4")
|
|
157
|
+
|
|
158
|
+
# 清理临时 cookie 文件
|
|
159
|
+
if cookie_file and os.path.exists(cookie_file):
|
|
160
|
+
os.unlink(cookie_file)
|
|
161
|
+
|
|
162
|
+
if not os.path.exists(video_path):
|
|
163
|
+
raise FileNotFoundError(f"视频文件未找到: {video_path}")
|
|
164
|
+
|
|
165
|
+
return video_path
|
|
166
|
+
|
|
167
|
+
def delete_video(self, video_path: str) -> str:
|
|
168
|
+
"""
|
|
169
|
+
删除视频文件
|
|
170
|
+
"""
|
|
171
|
+
if os.path.exists(video_path):
|
|
172
|
+
os.remove(video_path)
|
|
173
|
+
return f"视频文件已删除: {video_path}"
|
|
174
|
+
else:
|
|
175
|
+
return f"视频文件未找到: {video_path}"
|
|
176
|
+
|
|
177
|
+
def download_subtitles(self, video_url: str, output_dir: str = None,
|
|
178
|
+
langs: List[str] = None) -> Optional[TranscriptResult]:
|
|
179
|
+
"""
|
|
180
|
+
尝试获取B站视频字幕
|
|
181
|
+
|
|
182
|
+
:param video_url: 视频链接
|
|
183
|
+
:param output_dir: 输出路径
|
|
184
|
+
:param langs: 优先语言列表
|
|
185
|
+
:return: TranscriptResult 或 None
|
|
186
|
+
"""
|
|
187
|
+
if output_dir is None:
|
|
188
|
+
output_dir = get_path_manager().downloads_dir
|
|
189
|
+
os.makedirs(output_dir, exist_ok=True)
|
|
190
|
+
|
|
191
|
+
if langs is None:
|
|
192
|
+
langs = ['zh-Hans', 'zh', 'zh-CN', 'ai-zh', 'en', 'en-US']
|
|
193
|
+
|
|
194
|
+
video_id = extract_video_id(video_url, "bilibili")
|
|
195
|
+
|
|
196
|
+
ydl_opts = {
|
|
197
|
+
'writesubtitles': True,
|
|
198
|
+
'writeautomaticsub': True,
|
|
199
|
+
'subtitleslangs': langs,
|
|
200
|
+
'subtitlesformat': 'srt/json3/best', # 支持多种格式
|
|
201
|
+
'skip_download': True,
|
|
202
|
+
'outtmpl': os.path.join(output_dir, f'{video_id}.%(ext)s'),
|
|
203
|
+
'quiet': True,
|
|
204
|
+
'http_headers': {
|
|
205
|
+
'Referer': 'https://www.bilibili.com',
|
|
206
|
+
'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36',
|
|
207
|
+
},
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
cookie_file = _apply_bilibili_cookie(ydl_opts)
|
|
211
|
+
|
|
212
|
+
try:
|
|
213
|
+
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
|
214
|
+
# extract_info 不需要下载文件,与 skip_download 保持一致
|
|
215
|
+
info = ydl.extract_info(video_url, download=False)
|
|
216
|
+
|
|
217
|
+
# 查找下载的字幕文件
|
|
218
|
+
subtitles = info.get('requested_subtitles') or {}
|
|
219
|
+
if not subtitles:
|
|
220
|
+
logger.info(f"B站视频 {video_id} 没有可用字幕")
|
|
221
|
+
return None
|
|
222
|
+
|
|
223
|
+
# 按优先级查找字幕
|
|
224
|
+
detected_lang = None
|
|
225
|
+
sub_info = None
|
|
226
|
+
for lang in langs:
|
|
227
|
+
if lang in subtitles:
|
|
228
|
+
detected_lang = lang
|
|
229
|
+
sub_info = subtitles[lang]
|
|
230
|
+
break
|
|
231
|
+
|
|
232
|
+
# 如果按优先级没找到,取第一个可用的(排除弹幕)
|
|
233
|
+
if not detected_lang:
|
|
234
|
+
for lang, info_item in subtitles.items():
|
|
235
|
+
if lang != 'danmaku': # 排除弹幕
|
|
236
|
+
detected_lang = lang
|
|
237
|
+
sub_info = info_item
|
|
238
|
+
break
|
|
239
|
+
|
|
240
|
+
if not sub_info:
|
|
241
|
+
logger.info(f"B站视频 {video_id} 没有可用字幕(排除弹幕)")
|
|
242
|
+
return None
|
|
243
|
+
|
|
244
|
+
# 检查是否有内嵌数据(yt-dlp 有时直接返回字幕内容)
|
|
245
|
+
if 'data' in sub_info and sub_info['data']:
|
|
246
|
+
logger.info(f"直接从返回数据解析字幕: {detected_lang}")
|
|
247
|
+
return self._parse_srt_content(sub_info['data'], detected_lang)
|
|
248
|
+
|
|
249
|
+
# 查找字幕文件
|
|
250
|
+
ext = sub_info.get('ext', 'srt')
|
|
251
|
+
subtitle_file = os.path.join(output_dir, f"{video_id}.{detected_lang}.{ext}")
|
|
252
|
+
|
|
253
|
+
if not os.path.exists(subtitle_file):
|
|
254
|
+
logger.info(f"字幕文件不存在: {subtitle_file}")
|
|
255
|
+
return None
|
|
256
|
+
|
|
257
|
+
# 根据格式解析字幕文件
|
|
258
|
+
if ext == 'json3':
|
|
259
|
+
return self._parse_json3_subtitle(subtitle_file, detected_lang)
|
|
260
|
+
else:
|
|
261
|
+
with open(subtitle_file, 'r', encoding='utf-8') as f:
|
|
262
|
+
return self._parse_srt_content(f.read(), detected_lang)
|
|
263
|
+
|
|
264
|
+
except Exception as e:
|
|
265
|
+
logger.warning(f"获取B站字幕失败: {e}")
|
|
266
|
+
return None
|
|
267
|
+
finally:
|
|
268
|
+
if cookie_file and os.path.exists(cookie_file):
|
|
269
|
+
os.unlink(cookie_file)
|
|
270
|
+
|
|
271
|
+
def _parse_srt_content(self, srt_content: str, language: str) -> Optional[TranscriptResult]:
|
|
272
|
+
"""
|
|
273
|
+
解析 SRT 格式字幕内容
|
|
274
|
+
|
|
275
|
+
:param srt_content: SRT 字幕文本内容
|
|
276
|
+
:param language: 语言代码
|
|
277
|
+
:return: TranscriptResult
|
|
278
|
+
"""
|
|
279
|
+
import re
|
|
280
|
+
try:
|
|
281
|
+
segments = []
|
|
282
|
+
# SRT 格式: 序号\n时间戳\n文本\n\n
|
|
283
|
+
pattern = r'(\d+)\n(\d{2}:\d{2}:\d{2},\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2},\d{3})\n(.*?)(?=\n\n|\n\d+\n|$)'
|
|
284
|
+
matches = re.findall(pattern, srt_content, re.DOTALL)
|
|
285
|
+
|
|
286
|
+
for match in matches:
|
|
287
|
+
idx, start_time, end_time, text = match
|
|
288
|
+
text = text.strip()
|
|
289
|
+
if not text:
|
|
290
|
+
continue
|
|
291
|
+
|
|
292
|
+
# 转换时间格式 00:00:00,000 -> 秒
|
|
293
|
+
def time_to_seconds(t):
|
|
294
|
+
parts = t.replace(',', '.').split(':')
|
|
295
|
+
return float(parts[0]) * 3600 + float(parts[1]) * 60 + float(parts[2])
|
|
296
|
+
|
|
297
|
+
segments.append(TranscriptSegment(
|
|
298
|
+
start=time_to_seconds(start_time),
|
|
299
|
+
end=time_to_seconds(end_time),
|
|
300
|
+
text=text
|
|
301
|
+
))
|
|
302
|
+
|
|
303
|
+
if not segments:
|
|
304
|
+
return None
|
|
305
|
+
|
|
306
|
+
full_text = ' '.join(seg.text for seg in segments)
|
|
307
|
+
logger.info(f"成功解析B站SRT字幕,共 {len(segments)} 段")
|
|
308
|
+
return TranscriptResult(
|
|
309
|
+
language=language,
|
|
310
|
+
full_text=full_text,
|
|
311
|
+
segments=segments,
|
|
312
|
+
raw={'source': 'bilibili_subtitle', 'format': 'srt'}
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
except Exception as e:
|
|
316
|
+
logger.warning(f"解析SRT字幕失败: {e}")
|
|
317
|
+
return None
|
|
318
|
+
|
|
319
|
+
def _parse_json3_subtitle(self, subtitle_file: str, language: str) -> Optional[TranscriptResult]:
|
|
320
|
+
"""
|
|
321
|
+
解析 json3 格式字幕文件
|
|
322
|
+
|
|
323
|
+
:param subtitle_file: 字幕文件路径
|
|
324
|
+
:param language: 语言代码
|
|
325
|
+
:return: TranscriptResult
|
|
326
|
+
"""
|
|
327
|
+
try:
|
|
328
|
+
with open(subtitle_file, 'r', encoding='utf-8') as f:
|
|
329
|
+
data = json.load(f)
|
|
330
|
+
|
|
331
|
+
segments = []
|
|
332
|
+
events = data.get('events', [])
|
|
333
|
+
|
|
334
|
+
for event in events:
|
|
335
|
+
# json3 格式中时间单位是毫秒
|
|
336
|
+
start_ms = event.get('tStartMs', 0)
|
|
337
|
+
duration_ms = event.get('dDurationMs', 0)
|
|
338
|
+
|
|
339
|
+
# 提取文本
|
|
340
|
+
segs = event.get('segs', [])
|
|
341
|
+
text = ''.join(seg.get('utf8', '') for seg in segs).strip()
|
|
342
|
+
|
|
343
|
+
if text: # 只添加非空文本
|
|
344
|
+
segments.append(TranscriptSegment(
|
|
345
|
+
start=start_ms / 1000.0,
|
|
346
|
+
end=(start_ms + duration_ms) / 1000.0,
|
|
347
|
+
text=text
|
|
348
|
+
))
|
|
349
|
+
|
|
350
|
+
if not segments:
|
|
351
|
+
return None
|
|
352
|
+
|
|
353
|
+
full_text = ' '.join(seg.text for seg in segments)
|
|
354
|
+
|
|
355
|
+
logger.info(f"成功解析B站字幕,共 {len(segments)} 段")
|
|
356
|
+
return TranscriptResult(
|
|
357
|
+
language=language,
|
|
358
|
+
full_text=full_text,
|
|
359
|
+
segments=segments,
|
|
360
|
+
raw={'source': 'bilibili_subtitle', 'file': subtitle_file}
|
|
361
|
+
)
|
|
362
|
+
|
|
363
|
+
except Exception as e:
|
|
364
|
+
logger.warning(f"解析字幕文件失败: {e}")
|
|
365
|
+
return None
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# def download():
|