parsehub 2.0.41__tar.gz → 2.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. {parsehub-2.0.41/src/parsehub.egg-info → parsehub-2.1.0}/PKG-INFO +20 -20
  2. {parsehub-2.0.41 → parsehub-2.1.0}/README.md +19 -19
  3. {parsehub-2.0.41 → parsehub-2.1.0}/pyproject.toml +1 -1
  4. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/config/config.py +0 -4
  5. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/base/base.py +2 -3
  6. parsehub-2.1.0/src/parsehub/parsers/base/ytdlp.py +549 -0
  7. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/bilibili.py +22 -21
  8. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/youtube.py +21 -15
  9. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/coolapk.py +2 -2
  10. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/kuaishou.py +2 -2
  11. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/pipix.py +2 -2
  12. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/threads.py +2 -2
  13. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/tiktok.py +3 -3
  14. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/twitter.py +2 -2
  15. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/weixin.py +2 -2
  16. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/utils/helpers.py +2 -0
  17. {parsehub-2.0.41 → parsehub-2.1.0/src/parsehub.egg-info}/PKG-INFO +20 -20
  18. parsehub-2.0.41/src/parsehub/parsers/base/ytdlp.py +0 -284
  19. {parsehub-2.0.41 → parsehub-2.1.0}/LICENSE +0 -0
  20. {parsehub-2.0.41 → parsehub-2.1.0}/setup.cfg +0 -0
  21. {parsehub-2.0.41 → parsehub-2.1.0}/src/__init__.py +0 -0
  22. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/__init__.py +0 -0
  23. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/cli.py +0 -0
  24. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/cli_config.py +0 -0
  25. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/config/__init__.py +0 -0
  26. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/errors.py +0 -0
  27. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/__init__.py +0 -0
  28. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/base/__init__.py +0 -0
  29. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/__init__.py +0 -0
  30. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/coolapk.py +0 -0
  31. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/douyin.py +0 -0
  32. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/facebook.py +0 -0
  33. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/instagram.py +0 -0
  34. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/kuaishou.py +0 -0
  35. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/pipix.py +0 -0
  36. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/snapchat.py +0 -0
  37. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/threads.py +0 -0
  38. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/tieba.py +0 -0
  39. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/tiktok.py +0 -0
  40. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/twitter.py +0 -0
  41. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/weibo.py +0 -0
  42. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/weixin.py +0 -0
  43. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/xhs.py +0 -0
  44. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/xiaoheihe.py +0 -0
  45. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/zuiyou.py +0 -0
  46. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/__init__.py +0 -0
  47. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/bilibili.py +0 -0
  48. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/douyin.py +0 -0
  49. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/instagram.py +0 -0
  50. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/tieba.py +0 -0
  51. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/weibo.py +0 -0
  52. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/xhs.py +0 -0
  53. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/xiaoheihe.py +0 -0
  54. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/zuiyou.py +0 -0
  55. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/types/__init__.py +0 -0
  56. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/types/callback.py +0 -0
  57. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/types/media_file.py +0 -0
  58. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/types/media_ref.py +0 -0
  59. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/types/platform.py +0 -0
  60. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/types/post.py +0 -0
  61. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/types/result.py +0 -0
  62. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/utils/downloader.py +0 -0
  63. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/utils/media_info.py +0 -0
  64. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub.egg-info/SOURCES.txt +0 -0
  65. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub.egg-info/dependency_links.txt +0 -0
  66. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub.egg-info/entry_points.txt +0 -0
  67. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub.egg-info/requires.txt +0 -0
  68. {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub.egg-info/top_level.txt +0 -0
  69. {parsehub-2.0.41 → parsehub-2.1.0}/test/test_cli.py +0 -0
  70. {parsehub-2.0.41 → parsehub-2.1.0}/test/test_cli_config.py +0 -0
  71. {parsehub-2.0.41 → parsehub-2.1.0}/test/test_core_offline.py +0 -0
  72. {parsehub-2.0.41 → parsehub-2.1.0}/test/test_downloader.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: parsehub
3
- Version: 2.0.41
3
+ Version: 2.1.0
4
4
  Summary: 轻量、异步、开箱即用的社交媒体聚合解析库
5
5
  Author-email: 梓澪 <zilingmio@gmail.com>
6
6
  License: MIT
@@ -71,25 +71,25 @@ Dynamic: license-file
71
71
  ## 🌐 支持平台
72
72
 
73
73
  | 平台 | 视频 | 图文 | 其他 |
74
- |-----------------|:--:|:-:|------|
75
- | **Twitter / X** | ✅ | ✅ | |
76
- | **Instagram** | ✅ | ✅ | |
77
- | **YouTube** | ✅ | | 🎵 音乐 |
78
- | **Facebook** | ✅ | | |
79
- | **Threads** | ✅ | ✅ | |
80
- | **Bilibili** | ✅ | | 📝 动态 |
81
- | **抖音** | ✅ | ✅ | ☀️日常 |
82
- | **TikTok** | ✅ | ✅ | |
83
- | **微博** | ✅ | ✅ | |
84
- | **小红书** | ✅ | ✅ | |
85
- | **贴吧** | ✅ | ✅ | |
86
- | **微信公众号** | | ✅ | |
87
- | **快手** | ✅ | | |
88
- | **酷安** | | ✅ | |
89
- | **皮皮虾** | ✅ | ✅ | |
90
- | **最右** | ✅ | ✅ | |
91
- | **小黑盒** | ✅ | ✅ | |
92
- | **Snapchat** | ✅ | | |
74
+ |-----------------|:--:|:--:|------|
75
+ | **Twitter / X** | ✅ | ✅ | 📝 文章 |
76
+ | **Instagram** | ✅ | ✅ | |
77
+ | **YouTube** | ✅ | | 🎵 音乐 |
78
+ | **Facebook** | ✅ | | |
79
+ | **Threads** | ✅ | ✅ | |
80
+ | **Bilibili** | ✅ | | 📝 动态 |
81
+ | **抖音** | ✅ | ✅ | ☀️日常 |
82
+ | **TikTok** | ✅ | ✅ | |
83
+ | **微博** | ✅ | ✅ | |
84
+ | **小红书** | ✅ | ✅ | |
85
+ | **贴吧** | ✅ | ✅ | |
86
+ | **微信公众号** | | ✅ | |
87
+ | **快手** | ✅ | | |
88
+ | **酷安** | | ✅ | |
89
+ | **皮皮虾** | ✅ | ✅ | |
90
+ | **最右** | ✅ | ✅ | |
91
+ | **小黑盒** | ✅ | ✅ | |
92
+ | **Snapchat** | ✅ | | |
93
93
 
94
94
  ## 📦 安装
95
95
 
@@ -29,25 +29,25 @@
29
29
  ## 🌐 支持平台
30
30
 
31
31
  | 平台 | 视频 | 图文 | 其他 |
32
- |-----------------|:--:|:-:|------|
33
- | **Twitter / X** | ✅ | ✅ | |
34
- | **Instagram** | ✅ | ✅ | |
35
- | **YouTube** | ✅ | | 🎵 音乐 |
36
- | **Facebook** | ✅ | | |
37
- | **Threads** | ✅ | ✅ | |
38
- | **Bilibili** | ✅ | | 📝 动态 |
39
- | **抖音** | ✅ | ✅ | ☀️日常 |
40
- | **TikTok** | ✅ | ✅ | |
41
- | **微博** | ✅ | ✅ | |
42
- | **小红书** | ✅ | ✅ | |
43
- | **贴吧** | ✅ | ✅ | |
44
- | **微信公众号** | | ✅ | |
45
- | **快手** | ✅ | | |
46
- | **酷安** | | ✅ | |
47
- | **皮皮虾** | ✅ | ✅ | |
48
- | **最右** | ✅ | ✅ | |
49
- | **小黑盒** | ✅ | ✅ | |
50
- | **Snapchat** | ✅ | | |
32
+ |-----------------|:--:|:--:|------|
33
+ | **Twitter / X** | ✅ | ✅ | 📝 文章 |
34
+ | **Instagram** | ✅ | ✅ | |
35
+ | **YouTube** | ✅ | | 🎵 音乐 |
36
+ | **Facebook** | ✅ | | |
37
+ | **Threads** | ✅ | ✅ | |
38
+ | **Bilibili** | ✅ | | 📝 动态 |
39
+ | **抖音** | ✅ | ✅ | ☀️日常 |
40
+ | **TikTok** | ✅ | ✅ | |
41
+ | **微博** | ✅ | ✅ | |
42
+ | **小红书** | ✅ | ✅ | |
43
+ | **贴吧** | ✅ | ✅ | |
44
+ | **微信公众号** | | ✅ | |
45
+ | **快手** | ✅ | | |
46
+ | **酷安** | | ✅ | |
47
+ | **皮皮虾** | ✅ | ✅ | |
48
+ | **最右** | ✅ | ✅ | |
49
+ | **小黑盒** | ✅ | ✅ | |
50
+ | **Snapchat** | ✅ | | |
51
51
 
52
52
  ## 📦 安装
53
53
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "parsehub"
3
- version = "2.0.41"
3
+ version = "2.1.0"
4
4
  description = "轻量、异步、开箱即用的社交媒体聚合解析库"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.12.0"
@@ -7,10 +7,6 @@ from pydantic import BaseModel, ConfigDict
7
7
  class _GlobalConfig(BaseModel):
8
8
  model_config = ConfigDict(validate_assignment=True)
9
9
 
10
- ua: str = (
11
- "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
12
- "AppleWebKit/537.36 (KHTML, like Gecko) Chrome/144.0.0.0 Safari/537.36"
13
- )
14
10
  default_save_dir: Path = Path(sys.argv[0]).parent / "downloads"
15
11
  """默认下载目录"""
16
12
 
@@ -8,10 +8,9 @@ from urllib.parse import parse_qs, urlencode, urlparse
8
8
  import httpx
9
9
 
10
10
  from ... import parsers
11
- from ...config.config import GlobalConfig
12
11
  from ...types import AnyParseResult, ParseError
13
12
  from ...types.platform import Platform
14
- from ...utils.helpers import SecretCookie, match_url
13
+ from ...utils.helpers import UA, SecretCookie, match_url
15
14
 
16
15
 
17
16
  class BaseParser(ABC):
@@ -116,7 +115,7 @@ class BaseParser(ABC):
116
115
  r = await client.get(
117
116
  url,
118
117
  follow_redirects=True,
119
- headers={"User-Agent": GlobalConfig.ua},
118
+ headers={"User-Agent": UA},
120
119
  )
121
120
  except (httpx.ReadTimeout, httpx.ConnectTimeout) as e:
122
121
  raise ParseError("获取原始链接超时") from e
@@ -0,0 +1,549 @@
1
+ import asyncio
2
+ import json
3
+ import os
4
+ import signal
5
+ import sys
6
+ import tempfile
7
+ from collections import deque
8
+ from collections.abc import Iterator
9
+ from contextlib import contextmanager
10
+ from dataclasses import dataclass
11
+ from pathlib import Path
12
+ from typing import Any, cast
13
+
14
+ from loguru import logger
15
+
16
+ from ...types import (
17
+ DownloadError,
18
+ DownloadResult,
19
+ ParseError,
20
+ ProgressCallback,
21
+ VideoFile,
22
+ VideoParseResult,
23
+ VideoRef,
24
+ )
25
+ from .base import BaseParser
26
+
27
+ # 用一个不会和 yt-dlp 普通日志冲突的前缀标记进度行,stdout/stderr 读取时只解析这类行。
28
+ PROGRESS_PREFIX = "__PARSEHUB_YTDLP_PROGRESS__"
29
+
30
+ # yt-dlp CLI 进度模板, download: 是 yt-dlp 的模板作用域前缀;
31
+ # 后续字段用 tab 分隔,便于还原成 progress_hooks 风格的 dict。
32
+ PROGRESS_TEMPLATE = (
33
+ f"download:{PROGRESS_PREFIX}"
34
+ "%(progress.status)s\t"
35
+ "%(progress.downloaded_bytes)s\t"
36
+ "%(progress.total_bytes)s\t"
37
+ "%(progress.total_bytes_estimate)s\t"
38
+ "%(progress.fragment_index)s\t"
39
+ "%(progress.fragment_count)s"
40
+ )
41
+
42
+ # 子进程失败时只保留尾部日志用于错误信息,避免长输出占用过多内存或污染异常文本。
43
+ TAIL_LINES = 100
44
+ TAIL_CHARS = 16_000
45
+
46
+
47
+ class MonotonicDownloadProgress:
48
+ def __init__(self, *, start: float = 0.0, end: float = 100.0, min_step: float = 1.0) -> None:
49
+ self.start = start
50
+ self.end = end
51
+ self.min_step = max(1, int(min_step))
52
+ self.current = int(start)
53
+
54
+ def update(self, d: dict[str, Any]) -> int | None:
55
+ status = d.get("status")
56
+
57
+ if status == "downloading":
58
+ percent = self._download_percent(d)
59
+ if percent is None:
60
+ return None
61
+
62
+ mapped = int(self.start + percent * (self.end - self.start) / 100)
63
+
64
+ if mapped >= self.current + self.min_step:
65
+ self.current = mapped
66
+ return self.current
67
+
68
+ elif status == "finished" and self.current < int(self.end):
69
+ self.current = int(self.end)
70
+ return self.current
71
+
72
+ return None
73
+
74
+ @staticmethod
75
+ def _download_percent(d: dict[str, Any]) -> float | None:
76
+ downloaded = d.get("downloaded_bytes") or 0
77
+ total = d.get("total_bytes") or d.get("total_bytes_estimate") or 0
78
+ if downloaded == total == 1024:
79
+ return None
80
+
81
+ if total > 0:
82
+ return min(downloaded / total * 100, 100)
83
+
84
+ # 分片下载有时没有稳定总大小,但有 frag 进度;作为兜底
85
+ frag_index = d.get("fragment_index")
86
+ frag_count = d.get("fragment_count")
87
+ if isinstance(frag_index, int | float) and isinstance(frag_count, int | float) and frag_count:
88
+ return min(float(frag_index) / float(frag_count) * 100, 100.0)
89
+
90
+ return None
91
+
92
+
93
+ def _yt_dlp_base_cmd() -> list[str]:
94
+ return [sys.executable, "-m", "yt_dlp"]
95
+
96
+
97
+ def _subprocess_kwargs() -> dict[str, Any]:
98
+ if os.name == "posix":
99
+ return {"start_new_session": True}
100
+ return {}
101
+
102
+
103
+ async def _terminate_process(proc: asyncio.subprocess.Process) -> None:
104
+ if proc.returncode is not None:
105
+ return
106
+
107
+ if os.name == "posix":
108
+ try:
109
+ os.killpg(os.getpgid(proc.pid), signal.SIGTERM)
110
+ except ProcessLookupError:
111
+ return
112
+ except Exception:
113
+ proc.terminate()
114
+ else:
115
+ proc.terminate()
116
+
117
+ try:
118
+ await asyncio.wait_for(proc.wait(), timeout=5)
119
+ return
120
+ except TimeoutError:
121
+ pass
122
+
123
+ if proc.returncode is not None:
124
+ return
125
+
126
+ if os.name == "posix":
127
+ try:
128
+ os.killpg(os.getpgid(proc.pid), signal.SIGKILL)
129
+ except ProcessLookupError:
130
+ return
131
+ except Exception:
132
+ proc.kill()
133
+ else:
134
+ proc.kill()
135
+ await proc.wait()
136
+
137
+
138
+ @contextmanager
139
+ def _temporary_text_file(content: str, *, suffix: str) -> Iterator[str]:
140
+ path: str | None = None
141
+ try:
142
+ with tempfile.NamedTemporaryFile("w", encoding="utf-8", delete=False, suffix=suffix) as f:
143
+ path = f.name
144
+ f.write(content)
145
+ try:
146
+ os.chmod(path, 0o600)
147
+ except OSError:
148
+ pass
149
+ yield path
150
+ finally:
151
+ if path:
152
+ try:
153
+ os.unlink(path)
154
+ except FileNotFoundError:
155
+ pass
156
+ except OSError as e:
157
+ logger.debug("删除 yt-dlp 临时文件失败: {}", e)
158
+
159
+
160
+ @contextmanager
161
+ def _materialize_cookie(cookie_text: str | None) -> Iterator[list[str]]:
162
+ if not cookie_text:
163
+ yield []
164
+ return
165
+
166
+ with _temporary_text_file(cookie_text, suffix=".cookies.txt") as path:
167
+ yield ["--cookies", path]
168
+
169
+
170
+ @contextmanager
171
+ def _materialize_info_json(info_json: dict[str, Any]) -> Iterator[str]:
172
+ content = json.dumps(info_json, ensure_ascii=False)
173
+ with _temporary_text_file(content, suffix=".info.json") as path:
174
+ yield path
175
+
176
+
177
+ def _format_tail(tail: deque[str]) -> str:
178
+ return "".join(tail)[-TAIL_CHARS:].strip()
179
+
180
+
181
+ def _tail_from_text(text: str) -> deque[str]:
182
+ return deque(text.splitlines(keepends=True)[-TAIL_LINES:], maxlen=TAIL_LINES)
183
+
184
+
185
+ def _ytdlp_error(returncode: int, stdout_tail: deque[str], stderr_tail: deque[str]) -> str:
186
+ detail = _format_tail(stderr_tail) or _format_tail(stdout_tail) or "未知错误"
187
+ return f"yt-dlp exited with code {returncode}: {detail}"
188
+
189
+
190
+ def _decode_output(data: bytes) -> str:
191
+ return data.decode("utf-8", errors="replace")
192
+
193
+
194
+ def _json_from_stdout(stdout: str) -> dict[str, Any]:
195
+ text = stdout.strip()
196
+ if not text:
197
+ raise RuntimeError("yt-dlp 未输出 JSON")
198
+
199
+ try:
200
+ return cast(dict[str, Any], json.loads(text))
201
+ except json.JSONDecodeError:
202
+ start = text.find("{")
203
+ end = text.rfind("}")
204
+ if start == -1 or end == -1 or end <= start:
205
+ raise
206
+ return cast(dict[str, Any], json.loads(text[start : end + 1]))
207
+
208
+
209
+ def _optional_number(value: str) -> int | float | None:
210
+ value = value.strip()
211
+ if not value or value in {"NA", "None", "none", "null"}:
212
+ return None
213
+ try:
214
+ number = float(value)
215
+ except ValueError:
216
+ return None
217
+ if number.is_integer():
218
+ return int(number)
219
+ return number
220
+
221
+
222
+ def _parse_progress_line(line: str) -> dict[str, Any] | None:
223
+ index = line.find(PROGRESS_PREFIX)
224
+ if index == -1:
225
+ return None
226
+
227
+ payload = line[index + len(PROGRESS_PREFIX) :].strip()
228
+ parts = payload.split("\t")
229
+ if len(parts) < 6:
230
+ return None
231
+
232
+ return {
233
+ "status": parts[0],
234
+ "downloaded_bytes": _optional_number(parts[1]),
235
+ "total_bytes": _optional_number(parts[2]),
236
+ "total_bytes_estimate": _optional_number(parts[3]),
237
+ "fragment_index": _optional_number(parts[4]),
238
+ "fragment_count": _optional_number(parts[5]),
239
+ }
240
+
241
+
242
+ async def _run_ytdlp_json(
243
+ url: str,
244
+ cli_args: list[str],
245
+ *,
246
+ proxy: str | None = None,
247
+ cookie_text: str | None = None,
248
+ ) -> dict[str, Any]:
249
+ with _materialize_cookie(cookie_text) as cookie_args:
250
+ argv = [
251
+ *_yt_dlp_base_cmd(),
252
+ *cli_args,
253
+ *cookie_args,
254
+ ]
255
+ if proxy:
256
+ argv.extend(["--proxy", proxy])
257
+ argv.append(url)
258
+
259
+ proc = await asyncio.create_subprocess_exec(
260
+ *argv,
261
+ stdout=asyncio.subprocess.PIPE,
262
+ stderr=asyncio.subprocess.PIPE,
263
+ **_subprocess_kwargs(),
264
+ )
265
+ try:
266
+ stdout, stderr = await proc.communicate()
267
+ except asyncio.CancelledError:
268
+ await _terminate_process(proc)
269
+ raise
270
+
271
+ stdout_text = _decode_output(stdout)
272
+ stderr_text = _decode_output(stderr)
273
+ if proc.returncode:
274
+ raise RuntimeError(_ytdlp_error(proc.returncode, _tail_from_text(stdout_text), _tail_from_text(stderr_text)))
275
+
276
+ try:
277
+ return _json_from_stdout(stdout_text)
278
+ except Exception as e:
279
+ detail = stderr_text.strip() or str(e)
280
+ raise RuntimeError(f"解析 yt-dlp JSON 失败: {detail}") from e
281
+
282
+
283
+ async def _read_ytdlp_stream(
284
+ stream: asyncio.StreamReader,
285
+ tail: deque[str],
286
+ progress: MonotonicDownloadProgress | None,
287
+ callback: ProgressCallback | None,
288
+ callback_args: tuple,
289
+ callback_kwargs: dict,
290
+ ) -> None:
291
+ while line := await stream.readline():
292
+ text = _decode_output(line)
293
+ progress_data = _parse_progress_line(text)
294
+ if progress_data and progress and callback:
295
+ count = progress.update(progress_data)
296
+ if count is not None:
297
+ await callback(count, 100, "bytes", *callback_args, **callback_kwargs)
298
+ continue
299
+ tail.append(text)
300
+
301
+
302
+ async def _run_ytdlp_download(
303
+ info_json: dict[str, Any],
304
+ cli_args: list[str],
305
+ *,
306
+ outtmpl: str,
307
+ connections: int,
308
+ proxy: str | None = None,
309
+ headers: dict | None = None,
310
+ callback: ProgressCallback | None = None,
311
+ callback_args: tuple = (),
312
+ callback_kwargs: dict | None = None,
313
+ ) -> None:
314
+ callback_kwargs = callback_kwargs or {}
315
+ stdout_tail: deque[str] = deque(maxlen=TAIL_LINES)
316
+ stderr_tail: deque[str] = deque(maxlen=TAIL_LINES)
317
+ progress = MonotonicDownloadProgress(start=0, end=99) if callback else None
318
+
319
+ with _materialize_info_json(info_json) as info_path:
320
+ argv = [*_yt_dlp_base_cmd(), *cli_args]
321
+ if callback:
322
+ argv = [arg for arg in argv if arg not in {"--quiet", "--no-progress"}]
323
+ argv.extend(["--newline", "--progress-template", PROGRESS_TEMPLATE])
324
+ argv.extend(["--load-info-json", info_path, "-o", outtmpl, "-N", str(connections)])
325
+ if proxy:
326
+ argv.extend(["--proxy", proxy])
327
+ for key, value in (headers or {}).items():
328
+ argv.extend(["--add-header", f"{key}: {value}"])
329
+
330
+ proc = await asyncio.create_subprocess_exec(
331
+ *argv,
332
+ stdout=asyncio.subprocess.PIPE,
333
+ stderr=asyncio.subprocess.PIPE,
334
+ **_subprocess_kwargs(),
335
+ )
336
+ if proc.stdout is None or proc.stderr is None:
337
+ await _terminate_process(proc)
338
+ raise RuntimeError("yt-dlp 子进程 stdout/stderr 未正确初始化")
339
+
340
+ stdout_task = asyncio.create_task(
341
+ _read_ytdlp_stream(proc.stdout, stdout_tail, progress, callback, callback_args, callback_kwargs)
342
+ )
343
+ stderr_task = asyncio.create_task(
344
+ _read_ytdlp_stream(proc.stderr, stderr_tail, progress, callback, callback_args, callback_kwargs)
345
+ )
346
+ wait_task = asyncio.create_task(proc.wait())
347
+
348
+ try:
349
+ returncode = await wait_task
350
+ await asyncio.gather(stdout_task, stderr_task)
351
+ except asyncio.CancelledError:
352
+ await _terminate_process(proc)
353
+ for task in (stdout_task, stderr_task, wait_task):
354
+ task.cancel()
355
+ await asyncio.gather(stdout_task, stderr_task, wait_task, return_exceptions=True)
356
+ raise
357
+ except Exception:
358
+ await _terminate_process(proc)
359
+ for task in (stdout_task, stderr_task, wait_task):
360
+ task.cancel()
361
+ await asyncio.gather(stdout_task, stderr_task, wait_task, return_exceptions=True)
362
+ raise
363
+
364
+ if returncode:
365
+ raise RuntimeError(_ytdlp_error(returncode, stdout_tail, stderr_tail))
366
+
367
+
368
+ class YtParser(BaseParser, register=False):
369
+ """yt-dlp解析器"""
370
+
371
+ async def _do_parse(self, raw_url: str) -> "YtVideoParseResult":
372
+ video_info = await self._parse(raw_url)
373
+ return self._video_parse_result_type(
374
+ dl=video_info,
375
+ title=video_info.title,
376
+ content=video_info.description,
377
+ video=VideoRef(
378
+ url=raw_url,
379
+ thumb_url=video_info.thumbnail,
380
+ width=video_info.width,
381
+ height=video_info.height,
382
+ duration=video_info.duration,
383
+ ),
384
+ )
385
+
386
+ @property
387
+ def _video_parse_result_type(self) -> type["YtVideoParseResult"]:
388
+ return YtVideoParseResult
389
+
390
+ async def _parse(self, url: str) -> "YtVideoInfo":
391
+ try:
392
+ dl = await asyncio.wait_for(self._extract_info(url), timeout=30)
393
+ except TimeoutError as e:
394
+ raise ParseError("解析视频信息超时") from e
395
+ except Exception as e:
396
+ raise ParseError(f"解析视频信息失败: {str(e)}") from e
397
+
398
+ if dl.get("_type") == "playlist":
399
+ entries = dl.get("entries") or []
400
+ if not entries:
401
+ raise ParseError("解析视频信息失败: playlist entries is empty")
402
+ dl = entries[0]
403
+ url = dl.get("webpage_url") or url
404
+ title = dl["title"]
405
+ duration = dl.get("duration", 0)
406
+ thumbnail = dl["thumbnail"]
407
+ description = dl["description"]
408
+ width = dl.get("width", 0)
409
+ height = dl.get("height", 0)
410
+ return YtVideoInfo(
411
+ title=title,
412
+ description=description,
413
+ thumbnail=thumbnail,
414
+ duration=duration,
415
+ url=url,
416
+ width=width,
417
+ height=height,
418
+ info_json=dl,
419
+ )
420
+
421
+ async def _extract_info(self, url: str) -> dict[str, Any]:
422
+ return await _run_ytdlp_json(
423
+ url,
424
+ self.cli_args,
425
+ proxy=self.proxy,
426
+ cookie_text=self.get_cookie_text(),
427
+ )
428
+
429
+ def get_cookie_text(self) -> str | None:
430
+ return None
431
+
432
+ @property
433
+ def cli_args(self) -> list[str]:
434
+ return [
435
+ "--quiet", # 不输出日志
436
+ "--no-progress", # 不输出下载进度
437
+ "--no-playlist",
438
+ "--dump-single-json",
439
+ "--no-download",
440
+ "--no-warnings",
441
+ ]
442
+
443
+
444
+ class YtVideoParseResult(VideoParseResult):
445
+ def __init__(
446
+ self,
447
+ dl: "YtVideoInfo",
448
+ title: str | None,
449
+ video: VideoRef | None = None,
450
+ content: str | None = None,
451
+ ):
452
+ """dl: yt-dlp解析结果"""
453
+ self.dl = dl
454
+ super().__init__(title=title, video=video, content=content)
455
+
456
+ @property
457
+ def cli_args(self) -> list[str]:
458
+ return [
459
+ "--quiet", # 不输出日志
460
+ "--no-progress", # 不输出下载进度
461
+ ]
462
+
463
+ async def _do_download(
464
+ self,
465
+ *,
466
+ output_dir: Path,
467
+ callback: ProgressCallback | None = None,
468
+ callback_args: tuple = (),
469
+ callback_kwargs: dict | None = None,
470
+ proxy: str | None = None,
471
+ headers: dict | None = None,
472
+ connections: int = 4,
473
+ ) -> "DownloadResult":
474
+ if callback_kwargs is None:
475
+ callback_kwargs = {}
476
+ output_dir_path = Path(output_dir)
477
+
478
+ cli_args = self.cli_args.copy()
479
+ outtmpl = f"{output_dir_path.joinpath(self.name)}.%(ext)s"
480
+
481
+ await self._run_download(
482
+ cli_args,
483
+ outtmpl=outtmpl,
484
+ connections=connections,
485
+ proxy=proxy,
486
+ headers=headers,
487
+ callback=callback,
488
+ callback_args=callback_args,
489
+ callback_kwargs=callback_kwargs,
490
+ )
491
+
492
+ v = [p for p in output_dir_path.glob(f"{self.name}.*") if p.is_file()]
493
+ if not v:
494
+ raise DownloadError("下载失败: 未找到下载后的视频文件")
495
+
496
+ if callback:
497
+ await callback(100, 100, "bytes", *callback_args, **callback_kwargs)
498
+
499
+ video_path = v[0]
500
+ return DownloadResult(
501
+ VideoFile(
502
+ path=str(video_path),
503
+ height=self.dl.height,
504
+ width=self.dl.width,
505
+ duration=self.dl.duration,
506
+ ),
507
+ output_dir,
508
+ )
509
+
510
+ async def _run_download(
511
+ self,
512
+ cli_args: list[str],
513
+ *,
514
+ outtmpl: str,
515
+ connections: int,
516
+ proxy: str | None = None,
517
+ headers: dict | None = None,
518
+ callback: ProgressCallback | None = None,
519
+ callback_args: tuple = (),
520
+ callback_kwargs: dict | None = None,
521
+ ) -> None:
522
+ try:
523
+ await _run_ytdlp_download(
524
+ self.dl.info_json,
525
+ cli_args,
526
+ outtmpl=outtmpl,
527
+ connections=connections,
528
+ proxy=proxy,
529
+ headers=headers,
530
+ callback=callback,
531
+ callback_args=callback_args,
532
+ callback_kwargs=callback_kwargs,
533
+ )
534
+ except Exception as e:
535
+ raise DownloadError(f"下载失败: {str(e)}") from e
536
+
537
+
538
+ @dataclass
539
+ class YtVideoInfo:
540
+ """raw_video_info: yt-dlp解析结果"""
541
+
542
+ title: str
543
+ description: str
544
+ thumbnail: str
545
+ url: str
546
+ info_json: dict[str, Any]
547
+ duration: int = 0
548
+ width: int = 0
549
+ height: int = 0