bilinote-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. bilinote_cli-0.1.0.dist-info/METADATA +205 -0
  2. bilinote_cli-0.1.0.dist-info/RECORD +89 -0
  3. bilinote_cli-0.1.0.dist-info/WHEEL +4 -0
  4. bilinote_cli-0.1.0.dist-info/entry_points.txt +2 -0
  5. bilinote_cli-0.1.0.dist-info/licenses/LICENSE +21 -0
  6. src/__init__.py +0 -0
  7. src/app/config_manager.py +287 -0
  8. src/app/core/__init__.py +0 -0
  9. src/app/decorators/__init__.py +0 -0
  10. src/app/decorators/timeit.py +16 -0
  11. src/app/downloaders/__init__.py +0 -0
  12. src/app/downloaders/base.py +52 -0
  13. src/app/downloaders/bilibili_downloader.py +365 -0
  14. src/app/downloaders/common.py +1 -0
  15. src/app/downloaders/douyin_downloader.py +294 -0
  16. src/app/downloaders/douyin_helper/abogus.py +635 -0
  17. src/app/downloaders/kuaishou_downloader.py +93 -0
  18. src/app/downloaders/kuaishou_helper/__init__.py +0 -0
  19. src/app/downloaders/kuaishou_helper/kuaishou.py +96 -0
  20. src/app/downloaders/local_downloader.py +137 -0
  21. src/app/downloaders/xiaoyuzhoufm_download.py +25 -0
  22. src/app/downloaders/youtube_downloader.py +121 -0
  23. src/app/downloaders/youtube_subtitle.py +98 -0
  24. src/app/enmus/exception.py +21 -0
  25. src/app/enmus/note_enums.py +7 -0
  26. src/app/enmus/task_status_enums.py +28 -0
  27. src/app/exceptions/__init__.py +0 -0
  28. src/app/exceptions/biz_exception.py +6 -0
  29. src/app/exceptions/note.py +9 -0
  30. src/app/exceptions/provider.py +12 -0
  31. src/app/gpt/__init__.py +0 -0
  32. src/app/gpt/base.py +17 -0
  33. src/app/gpt/gpt_factory.py +13 -0
  34. src/app/gpt/prompt.py +77 -0
  35. src/app/gpt/prompt_builder.py +114 -0
  36. src/app/gpt/provider/OpenAI_compatible_provider.py +31 -0
  37. src/app/gpt/request_chunker.py +161 -0
  38. src/app/gpt/universal_gpt.py +348 -0
  39. src/app/gpt/utils.py +4 -0
  40. src/app/models/__init__.py +0 -0
  41. src/app/models/audio_model.py +15 -0
  42. src/app/models/gpt_model.py +19 -0
  43. src/app/models/model_config.py +16 -0
  44. src/app/models/notes_model.py +12 -0
  45. src/app/models/pipeline_model.py +27 -0
  46. src/app/models/provide_model.py +16 -0
  47. src/app/models/transcriber_model.py +16 -0
  48. src/app/models/video_record.py +0 -0
  49. src/app/secret_manager.py +116 -0
  50. src/app/services/__init__.py +0 -0
  51. src/app/services/batch_processor.py +343 -0
  52. src/app/services/cache/__init__.py +0 -0
  53. src/app/services/cache/task_cache.py +122 -0
  54. src/app/services/constant.py +17 -0
  55. src/app/services/note.py +141 -0
  56. src/app/services/pipeline/__init__.py +0 -0
  57. src/app/services/pipeline/ai_processor.py +98 -0
  58. src/app/services/pipeline/preparer.py +264 -0
  59. src/app/services/postprocessing.py +54 -0
  60. src/app/services/searcher.py +113 -0
  61. src/app/services/serial_executor.py +17 -0
  62. src/app/transcriber/__init__.py +0 -0
  63. src/app/transcriber/base.py +14 -0
  64. src/app/transcriber/bcut.py +245 -0
  65. src/app/transcriber/groq.py +76 -0
  66. src/app/transcriber/kuaishou.py +109 -0
  67. src/app/transcriber/transcriber_provider.py +134 -0
  68. src/app/transcriber/whisper_cpp.py +120 -0
  69. src/app/utils/cookie_helper.py +18 -0
  70. src/app/utils/env_checker.py +0 -0
  71. src/app/utils/file_cleanup.py +36 -0
  72. src/app/utils/logger.py +37 -0
  73. src/app/utils/note_helper.py +66 -0
  74. src/app/utils/path_helper.py +150 -0
  75. src/app/utils/screenshot_marker.py +13 -0
  76. src/app/utils/status_code.py +12 -0
  77. src/app/utils/url_parser.py +81 -0
  78. src/app/utils/video_helper.py +70 -0
  79. src/app/utils/video_reader.py +188 -0
  80. src/app/validators/__init__.py +0 -0
  81. src/app/validators/video_url_validator.py +38 -0
  82. src/cli.py +613 -0
  83. src/config/__init__.py +0 -0
  84. src/config/config.yaml.example +37 -0
  85. src/config/dev_config.json +6 -0
  86. src/config/model_config_manager.py +218 -0
  87. src/config/models.json +84 -0
  88. src/config/transcriber.json +29 -0
  89. src/ffmpeg_helper.py +49 -0
@@ -0,0 +1,52 @@
1
+ import enum
2
+
3
+ from abc import ABC, abstractmethod
4
+ from typing import Optional, Union
5
+
6
+ from app.enmus.note_enums import DownloadQuality
7
+ from app.models.notes_model import AudioDownloadResult
8
+ from app.models.transcriber_model import TranscriptResult
9
+ from os import getenv
10
+ QUALITY_MAP = {
11
+ "fast": "32",
12
+ "medium": "64",
13
+ "slow": "128"
14
+ }
15
+
16
+
17
+ class Downloader(ABC):
18
+ def __init__(self):
19
+ #TODO 需要修改为可配置
20
+ self.quality = QUALITY_MAP.get('fast')
21
+ self.cache_data=getenv('DATA_DIR')
22
+
23
+ @abstractmethod
24
+ def download(self, video_url: str, output_dir: str = None,
25
+ quality: DownloadQuality = "fast", need_video: Optional[bool] = False,
26
+ skip_download: bool = False) -> AudioDownloadResult:
27
+ '''
28
+
29
+ :param need_video:
30
+ :param video_url: 资源链接
31
+ :param output_dir: 输出路径 默认根目录data
32
+ :param quality: 音频质量 fast | medium | slow
33
+ :return:返回一个 AudioDownloadResult 类
34
+ '''
35
+ pass
36
+
37
+ @staticmethod
38
+ def download_video(self, video_url: str,
39
+ output_dir: Union[str, None] = None) -> str:
40
+ pass
41
+
42
+ def download_subtitles(self, video_url: str, output_dir: str = None,
43
+ langs: list = None) -> Optional[TranscriptResult]:
44
+ '''
45
+ 尝试获取平台字幕(人工字幕或自动生成字幕)
46
+
47
+ :param video_url: 视频链接
48
+ :param output_dir: 输出路径
49
+ :param langs: 优先语言列表,如 ['zh-Hans', 'zh', 'en']
50
+ :return: TranscriptResult 或 None(无字幕时)
51
+ '''
52
+ return None
@@ -0,0 +1,365 @@
1
+ import os
2
+ import json
3
+ import logging
4
+ import tempfile
5
+ from abc import ABC
6
+ from typing import Union, Optional, List
7
+ from pathlib import Path
8
+
9
+ import yt_dlp
10
+
11
+ from app.downloaders.base import Downloader, DownloadQuality, QUALITY_MAP
12
+ from app.models.notes_model import AudioDownloadResult
13
+ from app.models.transcriber_model import TranscriptResult, TranscriptSegment
14
+ from app.utils.path_helper import get_path_manager
15
+ from app.utils.url_parser import extract_video_id
16
+ from app.utils.cookie_helper import get_cookie
17
+
18
+ logger = logging.getLogger(__name__)
19
+
20
+
21
+ def _cookie_string_to_file(cookie_str: str, domain: str = ".bilibili.com") -> str:
22
+ """将 cookie 字符串转换为 Netscape 格式的临时文件,供 yt-dlp 使用"""
23
+ tmp_fd, tmp_path = tempfile.mkstemp(suffix=".txt", prefix="bili_cookie_")
24
+ # 使用一个足够远的未来时间戳作为 cookie 过期时间
25
+ expires = "2147483647"
26
+ with os.fdopen(tmp_fd, 'w', encoding='utf-8') as f:
27
+ f.write("# Netscape HTTP Cookie File\n")
28
+ for item in cookie_str.split(';'):
29
+ item = item.strip()
30
+ if '=' in item:
31
+ name, value = item.split('=', 1)
32
+ name = name.strip()
33
+ value = value.strip()
34
+ if name and value:
35
+ f.write(f"{domain}\tTRUE\t/\tFALSE\t{expires}\t{name}\t{value}\n")
36
+ return tmp_path
37
+
38
+
39
+ def _apply_bilibili_cookie(ydl_opts: dict):
40
+ """统一为 yt-dlp 选项添加 B站 cookie 支持"""
41
+ cookie_str = get_cookie('bilibili')
42
+ if cookie_str:
43
+ cookie_file = _cookie_string_to_file(cookie_str)
44
+ ydl_opts['cookiefile'] = cookie_file
45
+ return cookie_file
46
+ return None
47
+
48
+
49
+ class BilibiliDownloader(Downloader, ABC):
50
+ def __init__(self):
51
+ super().__init__()
52
+
53
+ def download(
54
+ self,
55
+ video_url: str,
56
+ output_dir: Union[str, None] = None,
57
+ quality: DownloadQuality = "fast",
58
+ need_video: Optional[bool] = False,
59
+ skip_download: bool = False,
60
+ ) -> AudioDownloadResult:
61
+ if output_dir is None:
62
+ output_dir = get_path_manager().downloads_dir
63
+ os.makedirs(output_dir, exist_ok=True)
64
+
65
+ output_path = os.path.join(output_dir, "%(id)s.%(ext)s")
66
+
67
+ ydl_opts = {
68
+ 'format': 'bestaudio[ext=m4a]/bestaudio/best',
69
+ 'outtmpl': output_path,
70
+ 'postprocessors': [
71
+ {
72
+ 'key': 'FFmpegExtractAudio',
73
+ 'preferredcodec': 'mp3',
74
+ 'preferredquality': '64',
75
+ }
76
+ ],
77
+ 'noplaylist': True,
78
+ 'quiet': False,
79
+ 'http_headers': {
80
+ 'Referer': 'https://www.bilibili.com',
81
+ 'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36',
82
+ },
83
+ }
84
+
85
+ cookie_file = _apply_bilibili_cookie(ydl_opts)
86
+
87
+ if skip_download:
88
+ ydl_opts['skip_download'] = True
89
+
90
+ with yt_dlp.YoutubeDL(ydl_opts) as ydl:
91
+ info = ydl.extract_info(video_url, download=not skip_download)
92
+ video_id = info.get("id")
93
+ title = info.get("title")
94
+ duration = info.get("duration", 0)
95
+ cover_url = info.get("thumbnail")
96
+ audio_path = os.path.join(output_dir, f"{video_id}.mp3")
97
+
98
+ # 清理临时 cookie 文件
99
+ if cookie_file and os.path.exists(cookie_file):
100
+ os.unlink(cookie_file)
101
+
102
+ if not skip_download and not os.path.exists(audio_path):
103
+ raise FileNotFoundError(f"音频下载失败,文件未生成: {audio_path}")
104
+
105
+ return AudioDownloadResult(
106
+ file_path=audio_path,
107
+ title=title,
108
+ duration=duration,
109
+ cover_url=cover_url,
110
+ platform="bilibili",
111
+ video_id=video_id,
112
+ raw_info=info,
113
+ video_path=None # ❗音频下载不包含视频路径
114
+ )
115
+
116
+ def download_video(
117
+ self,
118
+ video_url: str,
119
+ output_dir: Union[str, None] = None,
120
+ ) -> str:
121
+ """
122
+ 下载视频,返回视频文件路径
123
+ """
124
+
125
+ if output_dir is None:
126
+ output_dir = get_path_manager().downloads_dir
127
+ os.makedirs(output_dir, exist_ok=True)
128
+ print("video_url",video_url)
129
+ video_id=extract_video_id(video_url, "bilibili")
130
+ video_path = os.path.join(output_dir, f"{video_id}.mp4")
131
+ if os.path.exists(video_path):
132
+ return video_path
133
+
134
+ # 检查是否已经存在
135
+
136
+
137
+ output_path = os.path.join(output_dir, "%(id)s.%(ext)s")
138
+
139
+ ydl_opts = {
140
+ 'format': 'bv*[ext=mp4]/bestvideo+bestaudio/best',
141
+ 'outtmpl': output_path,
142
+ 'noplaylist': True,
143
+ 'quiet': False,
144
+ 'merge_output_format': 'mp4', # 确保合并成 mp4
145
+ 'http_headers': {
146
+ 'Referer': 'https://www.bilibili.com',
147
+ 'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36',
148
+ },
149
+ }
150
+
151
+ cookie_file = _apply_bilibili_cookie(ydl_opts)
152
+
153
+ with yt_dlp.YoutubeDL(ydl_opts) as ydl:
154
+ info = ydl.extract_info(video_url, download=True)
155
+ video_id = info.get("id")
156
+ video_path = os.path.join(output_dir, f"{video_id}.mp4")
157
+
158
+ # 清理临时 cookie 文件
159
+ if cookie_file and os.path.exists(cookie_file):
160
+ os.unlink(cookie_file)
161
+
162
+ if not os.path.exists(video_path):
163
+ raise FileNotFoundError(f"视频文件未找到: {video_path}")
164
+
165
+ return video_path
166
+
167
+ def delete_video(self, video_path: str) -> str:
168
+ """
169
+ 删除视频文件
170
+ """
171
+ if os.path.exists(video_path):
172
+ os.remove(video_path)
173
+ return f"视频文件已删除: {video_path}"
174
+ else:
175
+ return f"视频文件未找到: {video_path}"
176
+
177
+ def download_subtitles(self, video_url: str, output_dir: str = None,
178
+ langs: List[str] = None) -> Optional[TranscriptResult]:
179
+ """
180
+ 尝试获取B站视频字幕
181
+
182
+ :param video_url: 视频链接
183
+ :param output_dir: 输出路径
184
+ :param langs: 优先语言列表
185
+ :return: TranscriptResult 或 None
186
+ """
187
+ if output_dir is None:
188
+ output_dir = get_path_manager().downloads_dir
189
+ os.makedirs(output_dir, exist_ok=True)
190
+
191
+ if langs is None:
192
+ langs = ['zh-Hans', 'zh', 'zh-CN', 'ai-zh', 'en', 'en-US']
193
+
194
+ video_id = extract_video_id(video_url, "bilibili")
195
+
196
+ ydl_opts = {
197
+ 'writesubtitles': True,
198
+ 'writeautomaticsub': True,
199
+ 'subtitleslangs': langs,
200
+ 'subtitlesformat': 'srt/json3/best', # 支持多种格式
201
+ 'skip_download': True,
202
+ 'outtmpl': os.path.join(output_dir, f'{video_id}.%(ext)s'),
203
+ 'quiet': True,
204
+ 'http_headers': {
205
+ 'Referer': 'https://www.bilibili.com',
206
+ 'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36',
207
+ },
208
+ }
209
+
210
+ cookie_file = _apply_bilibili_cookie(ydl_opts)
211
+
212
+ try:
213
+ with yt_dlp.YoutubeDL(ydl_opts) as ydl:
214
+ # extract_info 不需要下载文件,与 skip_download 保持一致
215
+ info = ydl.extract_info(video_url, download=False)
216
+
217
+ # 查找下载的字幕文件
218
+ subtitles = info.get('requested_subtitles') or {}
219
+ if not subtitles:
220
+ logger.info(f"B站视频 {video_id} 没有可用字幕")
221
+ return None
222
+
223
+ # 按优先级查找字幕
224
+ detected_lang = None
225
+ sub_info = None
226
+ for lang in langs:
227
+ if lang in subtitles:
228
+ detected_lang = lang
229
+ sub_info = subtitles[lang]
230
+ break
231
+
232
+ # 如果按优先级没找到,取第一个可用的(排除弹幕)
233
+ if not detected_lang:
234
+ for lang, info_item in subtitles.items():
235
+ if lang != 'danmaku': # 排除弹幕
236
+ detected_lang = lang
237
+ sub_info = info_item
238
+ break
239
+
240
+ if not sub_info:
241
+ logger.info(f"B站视频 {video_id} 没有可用字幕(排除弹幕)")
242
+ return None
243
+
244
+ # 检查是否有内嵌数据(yt-dlp 有时直接返回字幕内容)
245
+ if 'data' in sub_info and sub_info['data']:
246
+ logger.info(f"直接从返回数据解析字幕: {detected_lang}")
247
+ return self._parse_srt_content(sub_info['data'], detected_lang)
248
+
249
+ # 查找字幕文件
250
+ ext = sub_info.get('ext', 'srt')
251
+ subtitle_file = os.path.join(output_dir, f"{video_id}.{detected_lang}.{ext}")
252
+
253
+ if not os.path.exists(subtitle_file):
254
+ logger.info(f"字幕文件不存在: {subtitle_file}")
255
+ return None
256
+
257
+ # 根据格式解析字幕文件
258
+ if ext == 'json3':
259
+ return self._parse_json3_subtitle(subtitle_file, detected_lang)
260
+ else:
261
+ with open(subtitle_file, 'r', encoding='utf-8') as f:
262
+ return self._parse_srt_content(f.read(), detected_lang)
263
+
264
+ except Exception as e:
265
+ logger.warning(f"获取B站字幕失败: {e}")
266
+ return None
267
+ finally:
268
+ if cookie_file and os.path.exists(cookie_file):
269
+ os.unlink(cookie_file)
270
+
271
+ def _parse_srt_content(self, srt_content: str, language: str) -> Optional[TranscriptResult]:
272
+ """
273
+ 解析 SRT 格式字幕内容
274
+
275
+ :param srt_content: SRT 字幕文本内容
276
+ :param language: 语言代码
277
+ :return: TranscriptResult
278
+ """
279
+ import re
280
+ try:
281
+ segments = []
282
+ # SRT 格式: 序号\n时间戳\n文本\n\n
283
+ pattern = r'(\d+)\n(\d{2}:\d{2}:\d{2},\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2},\d{3})\n(.*?)(?=\n\n|\n\d+\n|$)'
284
+ matches = re.findall(pattern, srt_content, re.DOTALL)
285
+
286
+ for match in matches:
287
+ idx, start_time, end_time, text = match
288
+ text = text.strip()
289
+ if not text:
290
+ continue
291
+
292
+ # 转换时间格式 00:00:00,000 -> 秒
293
+ def time_to_seconds(t):
294
+ parts = t.replace(',', '.').split(':')
295
+ return float(parts[0]) * 3600 + float(parts[1]) * 60 + float(parts[2])
296
+
297
+ segments.append(TranscriptSegment(
298
+ start=time_to_seconds(start_time),
299
+ end=time_to_seconds(end_time),
300
+ text=text
301
+ ))
302
+
303
+ if not segments:
304
+ return None
305
+
306
+ full_text = ' '.join(seg.text for seg in segments)
307
+ logger.info(f"成功解析B站SRT字幕,共 {len(segments)} 段")
308
+ return TranscriptResult(
309
+ language=language,
310
+ full_text=full_text,
311
+ segments=segments,
312
+ raw={'source': 'bilibili_subtitle', 'format': 'srt'}
313
+ )
314
+
315
+ except Exception as e:
316
+ logger.warning(f"解析SRT字幕失败: {e}")
317
+ return None
318
+
319
+ def _parse_json3_subtitle(self, subtitle_file: str, language: str) -> Optional[TranscriptResult]:
320
+ """
321
+ 解析 json3 格式字幕文件
322
+
323
+ :param subtitle_file: 字幕文件路径
324
+ :param language: 语言代码
325
+ :return: TranscriptResult
326
+ """
327
+ try:
328
+ with open(subtitle_file, 'r', encoding='utf-8') as f:
329
+ data = json.load(f)
330
+
331
+ segments = []
332
+ events = data.get('events', [])
333
+
334
+ for event in events:
335
+ # json3 格式中时间单位是毫秒
336
+ start_ms = event.get('tStartMs', 0)
337
+ duration_ms = event.get('dDurationMs', 0)
338
+
339
+ # 提取文本
340
+ segs = event.get('segs', [])
341
+ text = ''.join(seg.get('utf8', '') for seg in segs).strip()
342
+
343
+ if text: # 只添加非空文本
344
+ segments.append(TranscriptSegment(
345
+ start=start_ms / 1000.0,
346
+ end=(start_ms + duration_ms) / 1000.0,
347
+ text=text
348
+ ))
349
+
350
+ if not segments:
351
+ return None
352
+
353
+ full_text = ' '.join(seg.text for seg in segments)
354
+
355
+ logger.info(f"成功解析B站字幕,共 {len(segments)} 段")
356
+ return TranscriptResult(
357
+ language=language,
358
+ full_text=full_text,
359
+ segments=segments,
360
+ raw={'source': 'bilibili_subtitle', 'file': subtitle_file}
361
+ )
362
+
363
+ except Exception as e:
364
+ logger.warning(f"解析字幕文件失败: {e}")
365
+ return None
@@ -0,0 +1 @@
1
+ # def download():