parsehub 2.0.41__tar.gz → 2.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {parsehub-2.0.41/src/parsehub.egg-info → parsehub-2.1.0}/PKG-INFO +20 -20
- {parsehub-2.0.41 → parsehub-2.1.0}/README.md +19 -19
- {parsehub-2.0.41 → parsehub-2.1.0}/pyproject.toml +1 -1
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/config/config.py +0 -4
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/base/base.py +2 -3
- parsehub-2.1.0/src/parsehub/parsers/base/ytdlp.py +549 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/bilibili.py +22 -21
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/youtube.py +21 -15
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/coolapk.py +2 -2
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/kuaishou.py +2 -2
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/pipix.py +2 -2
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/threads.py +2 -2
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/tiktok.py +3 -3
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/twitter.py +2 -2
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/weixin.py +2 -2
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/utils/helpers.py +2 -0
- {parsehub-2.0.41 → parsehub-2.1.0/src/parsehub.egg-info}/PKG-INFO +20 -20
- parsehub-2.0.41/src/parsehub/parsers/base/ytdlp.py +0 -284
- {parsehub-2.0.41 → parsehub-2.1.0}/LICENSE +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/setup.cfg +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/__init__.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/__init__.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/cli.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/cli_config.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/config/__init__.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/errors.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/__init__.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/base/__init__.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/__init__.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/coolapk.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/douyin.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/facebook.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/instagram.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/kuaishou.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/pipix.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/snapchat.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/threads.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/tieba.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/tiktok.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/twitter.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/weibo.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/weixin.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/xhs.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/xiaoheihe.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/parsers/parser/zuiyou.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/__init__.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/bilibili.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/douyin.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/instagram.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/tieba.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/weibo.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/xhs.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/xiaoheihe.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/provider_api/zuiyou.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/types/__init__.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/types/callback.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/types/media_file.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/types/media_ref.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/types/platform.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/types/post.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/types/result.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/utils/downloader.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub/utils/media_info.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub.egg-info/SOURCES.txt +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub.egg-info/dependency_links.txt +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub.egg-info/entry_points.txt +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub.egg-info/requires.txt +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/src/parsehub.egg-info/top_level.txt +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/test/test_cli.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/test/test_cli_config.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/test/test_core_offline.py +0 -0
- {parsehub-2.0.41 → parsehub-2.1.0}/test/test_downloader.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: parsehub
|
|
3
|
-
Version: 2.0
|
|
3
|
+
Version: 2.1.0
|
|
4
4
|
Summary: 轻量、异步、开箱即用的社交媒体聚合解析库
|
|
5
5
|
Author-email: 梓澪 <zilingmio@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -71,25 +71,25 @@ Dynamic: license-file
|
|
|
71
71
|
## 🌐 支持平台
|
|
72
72
|
|
|
73
73
|
| 平台 | 视频 | 图文 | 其他 |
|
|
74
|
-
|
|
75
|
-
| **Twitter / X** | ✅ | ✅ |
|
|
76
|
-
| **Instagram** | ✅ | ✅
|
|
77
|
-
| **YouTube** | ✅ |
|
|
78
|
-
| **Facebook** | ✅ |
|
|
79
|
-
| **Threads** | ✅ | ✅
|
|
80
|
-
| **Bilibili** | ✅ |
|
|
81
|
-
| **抖音** | ✅ | ✅
|
|
82
|
-
| **TikTok** | ✅ | ✅
|
|
83
|
-
| **微博** | ✅ | ✅
|
|
84
|
-
| **小红书** | ✅ | ✅
|
|
85
|
-
| **贴吧** | ✅ | ✅
|
|
86
|
-
| **微信公众号** | | ✅
|
|
87
|
-
| **快手** | ✅ |
|
|
88
|
-
| **酷安** | | ✅
|
|
89
|
-
| **皮皮虾** | ✅ | ✅
|
|
90
|
-
| **最右** | ✅ | ✅
|
|
91
|
-
| **小黑盒** | ✅ | ✅
|
|
92
|
-
| **Snapchat** | ✅ |
|
|
74
|
+
|-----------------|:--:|:--:|------|
|
|
75
|
+
| **Twitter / X** | ✅ | ✅ | 📝 文章 |
|
|
76
|
+
| **Instagram** | ✅ | ✅ | |
|
|
77
|
+
| **YouTube** | ✅ | | 🎵 音乐 |
|
|
78
|
+
| **Facebook** | ✅ | | |
|
|
79
|
+
| **Threads** | ✅ | ✅ | |
|
|
80
|
+
| **Bilibili** | ✅ | | 📝 动态 |
|
|
81
|
+
| **抖音** | ✅ | ✅ | ☀️日常 |
|
|
82
|
+
| **TikTok** | ✅ | ✅ | |
|
|
83
|
+
| **微博** | ✅ | ✅ | |
|
|
84
|
+
| **小红书** | ✅ | ✅ | |
|
|
85
|
+
| **贴吧** | ✅ | ✅ | |
|
|
86
|
+
| **微信公众号** | | ✅ | |
|
|
87
|
+
| **快手** | ✅ | | |
|
|
88
|
+
| **酷安** | | ✅ | |
|
|
89
|
+
| **皮皮虾** | ✅ | ✅ | |
|
|
90
|
+
| **最右** | ✅ | ✅ | |
|
|
91
|
+
| **小黑盒** | ✅ | ✅ | |
|
|
92
|
+
| **Snapchat** | ✅ | | |
|
|
93
93
|
|
|
94
94
|
## 📦 安装
|
|
95
95
|
|
|
@@ -29,25 +29,25 @@
|
|
|
29
29
|
## 🌐 支持平台
|
|
30
30
|
|
|
31
31
|
| 平台 | 视频 | 图文 | 其他 |
|
|
32
|
-
|
|
33
|
-
| **Twitter / X** | ✅ | ✅ |
|
|
34
|
-
| **Instagram** | ✅ | ✅
|
|
35
|
-
| **YouTube** | ✅ |
|
|
36
|
-
| **Facebook** | ✅ |
|
|
37
|
-
| **Threads** | ✅ | ✅
|
|
38
|
-
| **Bilibili** | ✅ |
|
|
39
|
-
| **抖音** | ✅ | ✅
|
|
40
|
-
| **TikTok** | ✅ | ✅
|
|
41
|
-
| **微博** | ✅ | ✅
|
|
42
|
-
| **小红书** | ✅ | ✅
|
|
43
|
-
| **贴吧** | ✅ | ✅
|
|
44
|
-
| **微信公众号** | | ✅
|
|
45
|
-
| **快手** | ✅ |
|
|
46
|
-
| **酷安** | | ✅
|
|
47
|
-
| **皮皮虾** | ✅ | ✅
|
|
48
|
-
| **最右** | ✅ | ✅
|
|
49
|
-
| **小黑盒** | ✅ | ✅
|
|
50
|
-
| **Snapchat** | ✅ |
|
|
32
|
+
|-----------------|:--:|:--:|------|
|
|
33
|
+
| **Twitter / X** | ✅ | ✅ | 📝 文章 |
|
|
34
|
+
| **Instagram** | ✅ | ✅ | |
|
|
35
|
+
| **YouTube** | ✅ | | 🎵 音乐 |
|
|
36
|
+
| **Facebook** | ✅ | | |
|
|
37
|
+
| **Threads** | ✅ | ✅ | |
|
|
38
|
+
| **Bilibili** | ✅ | | 📝 动态 |
|
|
39
|
+
| **抖音** | ✅ | ✅ | ☀️日常 |
|
|
40
|
+
| **TikTok** | ✅ | ✅ | |
|
|
41
|
+
| **微博** | ✅ | ✅ | |
|
|
42
|
+
| **小红书** | ✅ | ✅ | |
|
|
43
|
+
| **贴吧** | ✅ | ✅ | |
|
|
44
|
+
| **微信公众号** | | ✅ | |
|
|
45
|
+
| **快手** | ✅ | | |
|
|
46
|
+
| **酷安** | | ✅ | |
|
|
47
|
+
| **皮皮虾** | ✅ | ✅ | |
|
|
48
|
+
| **最右** | ✅ | ✅ | |
|
|
49
|
+
| **小黑盒** | ✅ | ✅ | |
|
|
50
|
+
| **Snapchat** | ✅ | | |
|
|
51
51
|
|
|
52
52
|
## 📦 安装
|
|
53
53
|
|
|
@@ -7,10 +7,6 @@ from pydantic import BaseModel, ConfigDict
|
|
|
7
7
|
class _GlobalConfig(BaseModel):
|
|
8
8
|
model_config = ConfigDict(validate_assignment=True)
|
|
9
9
|
|
|
10
|
-
ua: str = (
|
|
11
|
-
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
|
12
|
-
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/144.0.0.0 Safari/537.36"
|
|
13
|
-
)
|
|
14
10
|
default_save_dir: Path = Path(sys.argv[0]).parent / "downloads"
|
|
15
11
|
"""默认下载目录"""
|
|
16
12
|
|
|
@@ -8,10 +8,9 @@ from urllib.parse import parse_qs, urlencode, urlparse
|
|
|
8
8
|
import httpx
|
|
9
9
|
|
|
10
10
|
from ... import parsers
|
|
11
|
-
from ...config.config import GlobalConfig
|
|
12
11
|
from ...types import AnyParseResult, ParseError
|
|
13
12
|
from ...types.platform import Platform
|
|
14
|
-
from ...utils.helpers import SecretCookie, match_url
|
|
13
|
+
from ...utils.helpers import UA, SecretCookie, match_url
|
|
15
14
|
|
|
16
15
|
|
|
17
16
|
class BaseParser(ABC):
|
|
@@ -116,7 +115,7 @@ class BaseParser(ABC):
|
|
|
116
115
|
r = await client.get(
|
|
117
116
|
url,
|
|
118
117
|
follow_redirects=True,
|
|
119
|
-
headers={"User-Agent":
|
|
118
|
+
headers={"User-Agent": UA},
|
|
120
119
|
)
|
|
121
120
|
except (httpx.ReadTimeout, httpx.ConnectTimeout) as e:
|
|
122
121
|
raise ParseError("获取原始链接超时") from e
|
|
@@ -0,0 +1,549 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
import json
|
|
3
|
+
import os
|
|
4
|
+
import signal
|
|
5
|
+
import sys
|
|
6
|
+
import tempfile
|
|
7
|
+
from collections import deque
|
|
8
|
+
from collections.abc import Iterator
|
|
9
|
+
from contextlib import contextmanager
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Any, cast
|
|
13
|
+
|
|
14
|
+
from loguru import logger
|
|
15
|
+
|
|
16
|
+
from ...types import (
|
|
17
|
+
DownloadError,
|
|
18
|
+
DownloadResult,
|
|
19
|
+
ParseError,
|
|
20
|
+
ProgressCallback,
|
|
21
|
+
VideoFile,
|
|
22
|
+
VideoParseResult,
|
|
23
|
+
VideoRef,
|
|
24
|
+
)
|
|
25
|
+
from .base import BaseParser
|
|
26
|
+
|
|
27
|
+
# 用一个不会和 yt-dlp 普通日志冲突的前缀标记进度行,stdout/stderr 读取时只解析这类行。
|
|
28
|
+
PROGRESS_PREFIX = "__PARSEHUB_YTDLP_PROGRESS__"
|
|
29
|
+
|
|
30
|
+
# yt-dlp CLI 进度模板, download: 是 yt-dlp 的模板作用域前缀;
|
|
31
|
+
# 后续字段用 tab 分隔,便于还原成 progress_hooks 风格的 dict。
|
|
32
|
+
PROGRESS_TEMPLATE = (
|
|
33
|
+
f"download:{PROGRESS_PREFIX}"
|
|
34
|
+
"%(progress.status)s\t"
|
|
35
|
+
"%(progress.downloaded_bytes)s\t"
|
|
36
|
+
"%(progress.total_bytes)s\t"
|
|
37
|
+
"%(progress.total_bytes_estimate)s\t"
|
|
38
|
+
"%(progress.fragment_index)s\t"
|
|
39
|
+
"%(progress.fragment_count)s"
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
# 子进程失败时只保留尾部日志用于错误信息,避免长输出占用过多内存或污染异常文本。
|
|
43
|
+
TAIL_LINES = 100
|
|
44
|
+
TAIL_CHARS = 16_000
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class MonotonicDownloadProgress:
|
|
48
|
+
def __init__(self, *, start: float = 0.0, end: float = 100.0, min_step: float = 1.0) -> None:
|
|
49
|
+
self.start = start
|
|
50
|
+
self.end = end
|
|
51
|
+
self.min_step = max(1, int(min_step))
|
|
52
|
+
self.current = int(start)
|
|
53
|
+
|
|
54
|
+
def update(self, d: dict[str, Any]) -> int | None:
|
|
55
|
+
status = d.get("status")
|
|
56
|
+
|
|
57
|
+
if status == "downloading":
|
|
58
|
+
percent = self._download_percent(d)
|
|
59
|
+
if percent is None:
|
|
60
|
+
return None
|
|
61
|
+
|
|
62
|
+
mapped = int(self.start + percent * (self.end - self.start) / 100)
|
|
63
|
+
|
|
64
|
+
if mapped >= self.current + self.min_step:
|
|
65
|
+
self.current = mapped
|
|
66
|
+
return self.current
|
|
67
|
+
|
|
68
|
+
elif status == "finished" and self.current < int(self.end):
|
|
69
|
+
self.current = int(self.end)
|
|
70
|
+
return self.current
|
|
71
|
+
|
|
72
|
+
return None
|
|
73
|
+
|
|
74
|
+
@staticmethod
|
|
75
|
+
def _download_percent(d: dict[str, Any]) -> float | None:
|
|
76
|
+
downloaded = d.get("downloaded_bytes") or 0
|
|
77
|
+
total = d.get("total_bytes") or d.get("total_bytes_estimate") or 0
|
|
78
|
+
if downloaded == total == 1024:
|
|
79
|
+
return None
|
|
80
|
+
|
|
81
|
+
if total > 0:
|
|
82
|
+
return min(downloaded / total * 100, 100)
|
|
83
|
+
|
|
84
|
+
# 分片下载有时没有稳定总大小,但有 frag 进度;作为兜底
|
|
85
|
+
frag_index = d.get("fragment_index")
|
|
86
|
+
frag_count = d.get("fragment_count")
|
|
87
|
+
if isinstance(frag_index, int | float) and isinstance(frag_count, int | float) and frag_count:
|
|
88
|
+
return min(float(frag_index) / float(frag_count) * 100, 100.0)
|
|
89
|
+
|
|
90
|
+
return None
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _yt_dlp_base_cmd() -> list[str]:
|
|
94
|
+
return [sys.executable, "-m", "yt_dlp"]
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _subprocess_kwargs() -> dict[str, Any]:
|
|
98
|
+
if os.name == "posix":
|
|
99
|
+
return {"start_new_session": True}
|
|
100
|
+
return {}
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
async def _terminate_process(proc: asyncio.subprocess.Process) -> None:
|
|
104
|
+
if proc.returncode is not None:
|
|
105
|
+
return
|
|
106
|
+
|
|
107
|
+
if os.name == "posix":
|
|
108
|
+
try:
|
|
109
|
+
os.killpg(os.getpgid(proc.pid), signal.SIGTERM)
|
|
110
|
+
except ProcessLookupError:
|
|
111
|
+
return
|
|
112
|
+
except Exception:
|
|
113
|
+
proc.terminate()
|
|
114
|
+
else:
|
|
115
|
+
proc.terminate()
|
|
116
|
+
|
|
117
|
+
try:
|
|
118
|
+
await asyncio.wait_for(proc.wait(), timeout=5)
|
|
119
|
+
return
|
|
120
|
+
except TimeoutError:
|
|
121
|
+
pass
|
|
122
|
+
|
|
123
|
+
if proc.returncode is not None:
|
|
124
|
+
return
|
|
125
|
+
|
|
126
|
+
if os.name == "posix":
|
|
127
|
+
try:
|
|
128
|
+
os.killpg(os.getpgid(proc.pid), signal.SIGKILL)
|
|
129
|
+
except ProcessLookupError:
|
|
130
|
+
return
|
|
131
|
+
except Exception:
|
|
132
|
+
proc.kill()
|
|
133
|
+
else:
|
|
134
|
+
proc.kill()
|
|
135
|
+
await proc.wait()
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
@contextmanager
|
|
139
|
+
def _temporary_text_file(content: str, *, suffix: str) -> Iterator[str]:
|
|
140
|
+
path: str | None = None
|
|
141
|
+
try:
|
|
142
|
+
with tempfile.NamedTemporaryFile("w", encoding="utf-8", delete=False, suffix=suffix) as f:
|
|
143
|
+
path = f.name
|
|
144
|
+
f.write(content)
|
|
145
|
+
try:
|
|
146
|
+
os.chmod(path, 0o600)
|
|
147
|
+
except OSError:
|
|
148
|
+
pass
|
|
149
|
+
yield path
|
|
150
|
+
finally:
|
|
151
|
+
if path:
|
|
152
|
+
try:
|
|
153
|
+
os.unlink(path)
|
|
154
|
+
except FileNotFoundError:
|
|
155
|
+
pass
|
|
156
|
+
except OSError as e:
|
|
157
|
+
logger.debug("删除 yt-dlp 临时文件失败: {}", e)
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
@contextmanager
|
|
161
|
+
def _materialize_cookie(cookie_text: str | None) -> Iterator[list[str]]:
|
|
162
|
+
if not cookie_text:
|
|
163
|
+
yield []
|
|
164
|
+
return
|
|
165
|
+
|
|
166
|
+
with _temporary_text_file(cookie_text, suffix=".cookies.txt") as path:
|
|
167
|
+
yield ["--cookies", path]
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
@contextmanager
|
|
171
|
+
def _materialize_info_json(info_json: dict[str, Any]) -> Iterator[str]:
|
|
172
|
+
content = json.dumps(info_json, ensure_ascii=False)
|
|
173
|
+
with _temporary_text_file(content, suffix=".info.json") as path:
|
|
174
|
+
yield path
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _format_tail(tail: deque[str]) -> str:
|
|
178
|
+
return "".join(tail)[-TAIL_CHARS:].strip()
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def _tail_from_text(text: str) -> deque[str]:
|
|
182
|
+
return deque(text.splitlines(keepends=True)[-TAIL_LINES:], maxlen=TAIL_LINES)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _ytdlp_error(returncode: int, stdout_tail: deque[str], stderr_tail: deque[str]) -> str:
|
|
186
|
+
detail = _format_tail(stderr_tail) or _format_tail(stdout_tail) or "未知错误"
|
|
187
|
+
return f"yt-dlp exited with code {returncode}: {detail}"
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _decode_output(data: bytes) -> str:
|
|
191
|
+
return data.decode("utf-8", errors="replace")
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _json_from_stdout(stdout: str) -> dict[str, Any]:
|
|
195
|
+
text = stdout.strip()
|
|
196
|
+
if not text:
|
|
197
|
+
raise RuntimeError("yt-dlp 未输出 JSON")
|
|
198
|
+
|
|
199
|
+
try:
|
|
200
|
+
return cast(dict[str, Any], json.loads(text))
|
|
201
|
+
except json.JSONDecodeError:
|
|
202
|
+
start = text.find("{")
|
|
203
|
+
end = text.rfind("}")
|
|
204
|
+
if start == -1 or end == -1 or end <= start:
|
|
205
|
+
raise
|
|
206
|
+
return cast(dict[str, Any], json.loads(text[start : end + 1]))
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _optional_number(value: str) -> int | float | None:
|
|
210
|
+
value = value.strip()
|
|
211
|
+
if not value or value in {"NA", "None", "none", "null"}:
|
|
212
|
+
return None
|
|
213
|
+
try:
|
|
214
|
+
number = float(value)
|
|
215
|
+
except ValueError:
|
|
216
|
+
return None
|
|
217
|
+
if number.is_integer():
|
|
218
|
+
return int(number)
|
|
219
|
+
return number
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _parse_progress_line(line: str) -> dict[str, Any] | None:
|
|
223
|
+
index = line.find(PROGRESS_PREFIX)
|
|
224
|
+
if index == -1:
|
|
225
|
+
return None
|
|
226
|
+
|
|
227
|
+
payload = line[index + len(PROGRESS_PREFIX) :].strip()
|
|
228
|
+
parts = payload.split("\t")
|
|
229
|
+
if len(parts) < 6:
|
|
230
|
+
return None
|
|
231
|
+
|
|
232
|
+
return {
|
|
233
|
+
"status": parts[0],
|
|
234
|
+
"downloaded_bytes": _optional_number(parts[1]),
|
|
235
|
+
"total_bytes": _optional_number(parts[2]),
|
|
236
|
+
"total_bytes_estimate": _optional_number(parts[3]),
|
|
237
|
+
"fragment_index": _optional_number(parts[4]),
|
|
238
|
+
"fragment_count": _optional_number(parts[5]),
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
async def _run_ytdlp_json(
|
|
243
|
+
url: str,
|
|
244
|
+
cli_args: list[str],
|
|
245
|
+
*,
|
|
246
|
+
proxy: str | None = None,
|
|
247
|
+
cookie_text: str | None = None,
|
|
248
|
+
) -> dict[str, Any]:
|
|
249
|
+
with _materialize_cookie(cookie_text) as cookie_args:
|
|
250
|
+
argv = [
|
|
251
|
+
*_yt_dlp_base_cmd(),
|
|
252
|
+
*cli_args,
|
|
253
|
+
*cookie_args,
|
|
254
|
+
]
|
|
255
|
+
if proxy:
|
|
256
|
+
argv.extend(["--proxy", proxy])
|
|
257
|
+
argv.append(url)
|
|
258
|
+
|
|
259
|
+
proc = await asyncio.create_subprocess_exec(
|
|
260
|
+
*argv,
|
|
261
|
+
stdout=asyncio.subprocess.PIPE,
|
|
262
|
+
stderr=asyncio.subprocess.PIPE,
|
|
263
|
+
**_subprocess_kwargs(),
|
|
264
|
+
)
|
|
265
|
+
try:
|
|
266
|
+
stdout, stderr = await proc.communicate()
|
|
267
|
+
except asyncio.CancelledError:
|
|
268
|
+
await _terminate_process(proc)
|
|
269
|
+
raise
|
|
270
|
+
|
|
271
|
+
stdout_text = _decode_output(stdout)
|
|
272
|
+
stderr_text = _decode_output(stderr)
|
|
273
|
+
if proc.returncode:
|
|
274
|
+
raise RuntimeError(_ytdlp_error(proc.returncode, _tail_from_text(stdout_text), _tail_from_text(stderr_text)))
|
|
275
|
+
|
|
276
|
+
try:
|
|
277
|
+
return _json_from_stdout(stdout_text)
|
|
278
|
+
except Exception as e:
|
|
279
|
+
detail = stderr_text.strip() or str(e)
|
|
280
|
+
raise RuntimeError(f"解析 yt-dlp JSON 失败: {detail}") from e
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
async def _read_ytdlp_stream(
|
|
284
|
+
stream: asyncio.StreamReader,
|
|
285
|
+
tail: deque[str],
|
|
286
|
+
progress: MonotonicDownloadProgress | None,
|
|
287
|
+
callback: ProgressCallback | None,
|
|
288
|
+
callback_args: tuple,
|
|
289
|
+
callback_kwargs: dict,
|
|
290
|
+
) -> None:
|
|
291
|
+
while line := await stream.readline():
|
|
292
|
+
text = _decode_output(line)
|
|
293
|
+
progress_data = _parse_progress_line(text)
|
|
294
|
+
if progress_data and progress and callback:
|
|
295
|
+
count = progress.update(progress_data)
|
|
296
|
+
if count is not None:
|
|
297
|
+
await callback(count, 100, "bytes", *callback_args, **callback_kwargs)
|
|
298
|
+
continue
|
|
299
|
+
tail.append(text)
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
async def _run_ytdlp_download(
|
|
303
|
+
info_json: dict[str, Any],
|
|
304
|
+
cli_args: list[str],
|
|
305
|
+
*,
|
|
306
|
+
outtmpl: str,
|
|
307
|
+
connections: int,
|
|
308
|
+
proxy: str | None = None,
|
|
309
|
+
headers: dict | None = None,
|
|
310
|
+
callback: ProgressCallback | None = None,
|
|
311
|
+
callback_args: tuple = (),
|
|
312
|
+
callback_kwargs: dict | None = None,
|
|
313
|
+
) -> None:
|
|
314
|
+
callback_kwargs = callback_kwargs or {}
|
|
315
|
+
stdout_tail: deque[str] = deque(maxlen=TAIL_LINES)
|
|
316
|
+
stderr_tail: deque[str] = deque(maxlen=TAIL_LINES)
|
|
317
|
+
progress = MonotonicDownloadProgress(start=0, end=99) if callback else None
|
|
318
|
+
|
|
319
|
+
with _materialize_info_json(info_json) as info_path:
|
|
320
|
+
argv = [*_yt_dlp_base_cmd(), *cli_args]
|
|
321
|
+
if callback:
|
|
322
|
+
argv = [arg for arg in argv if arg not in {"--quiet", "--no-progress"}]
|
|
323
|
+
argv.extend(["--newline", "--progress-template", PROGRESS_TEMPLATE])
|
|
324
|
+
argv.extend(["--load-info-json", info_path, "-o", outtmpl, "-N", str(connections)])
|
|
325
|
+
if proxy:
|
|
326
|
+
argv.extend(["--proxy", proxy])
|
|
327
|
+
for key, value in (headers or {}).items():
|
|
328
|
+
argv.extend(["--add-header", f"{key}: {value}"])
|
|
329
|
+
|
|
330
|
+
proc = await asyncio.create_subprocess_exec(
|
|
331
|
+
*argv,
|
|
332
|
+
stdout=asyncio.subprocess.PIPE,
|
|
333
|
+
stderr=asyncio.subprocess.PIPE,
|
|
334
|
+
**_subprocess_kwargs(),
|
|
335
|
+
)
|
|
336
|
+
if proc.stdout is None or proc.stderr is None:
|
|
337
|
+
await _terminate_process(proc)
|
|
338
|
+
raise RuntimeError("yt-dlp 子进程 stdout/stderr 未正确初始化")
|
|
339
|
+
|
|
340
|
+
stdout_task = asyncio.create_task(
|
|
341
|
+
_read_ytdlp_stream(proc.stdout, stdout_tail, progress, callback, callback_args, callback_kwargs)
|
|
342
|
+
)
|
|
343
|
+
stderr_task = asyncio.create_task(
|
|
344
|
+
_read_ytdlp_stream(proc.stderr, stderr_tail, progress, callback, callback_args, callback_kwargs)
|
|
345
|
+
)
|
|
346
|
+
wait_task = asyncio.create_task(proc.wait())
|
|
347
|
+
|
|
348
|
+
try:
|
|
349
|
+
returncode = await wait_task
|
|
350
|
+
await asyncio.gather(stdout_task, stderr_task)
|
|
351
|
+
except asyncio.CancelledError:
|
|
352
|
+
await _terminate_process(proc)
|
|
353
|
+
for task in (stdout_task, stderr_task, wait_task):
|
|
354
|
+
task.cancel()
|
|
355
|
+
await asyncio.gather(stdout_task, stderr_task, wait_task, return_exceptions=True)
|
|
356
|
+
raise
|
|
357
|
+
except Exception:
|
|
358
|
+
await _terminate_process(proc)
|
|
359
|
+
for task in (stdout_task, stderr_task, wait_task):
|
|
360
|
+
task.cancel()
|
|
361
|
+
await asyncio.gather(stdout_task, stderr_task, wait_task, return_exceptions=True)
|
|
362
|
+
raise
|
|
363
|
+
|
|
364
|
+
if returncode:
|
|
365
|
+
raise RuntimeError(_ytdlp_error(returncode, stdout_tail, stderr_tail))
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
class YtParser(BaseParser, register=False):
|
|
369
|
+
"""yt-dlp解析器"""
|
|
370
|
+
|
|
371
|
+
async def _do_parse(self, raw_url: str) -> "YtVideoParseResult":
|
|
372
|
+
video_info = await self._parse(raw_url)
|
|
373
|
+
return self._video_parse_result_type(
|
|
374
|
+
dl=video_info,
|
|
375
|
+
title=video_info.title,
|
|
376
|
+
content=video_info.description,
|
|
377
|
+
video=VideoRef(
|
|
378
|
+
url=raw_url,
|
|
379
|
+
thumb_url=video_info.thumbnail,
|
|
380
|
+
width=video_info.width,
|
|
381
|
+
height=video_info.height,
|
|
382
|
+
duration=video_info.duration,
|
|
383
|
+
),
|
|
384
|
+
)
|
|
385
|
+
|
|
386
|
+
@property
|
|
387
|
+
def _video_parse_result_type(self) -> type["YtVideoParseResult"]:
|
|
388
|
+
return YtVideoParseResult
|
|
389
|
+
|
|
390
|
+
async def _parse(self, url: str) -> "YtVideoInfo":
|
|
391
|
+
try:
|
|
392
|
+
dl = await asyncio.wait_for(self._extract_info(url), timeout=30)
|
|
393
|
+
except TimeoutError as e:
|
|
394
|
+
raise ParseError("解析视频信息超时") from e
|
|
395
|
+
except Exception as e:
|
|
396
|
+
raise ParseError(f"解析视频信息失败: {str(e)}") from e
|
|
397
|
+
|
|
398
|
+
if dl.get("_type") == "playlist":
|
|
399
|
+
entries = dl.get("entries") or []
|
|
400
|
+
if not entries:
|
|
401
|
+
raise ParseError("解析视频信息失败: playlist entries is empty")
|
|
402
|
+
dl = entries[0]
|
|
403
|
+
url = dl.get("webpage_url") or url
|
|
404
|
+
title = dl["title"]
|
|
405
|
+
duration = dl.get("duration", 0)
|
|
406
|
+
thumbnail = dl["thumbnail"]
|
|
407
|
+
description = dl["description"]
|
|
408
|
+
width = dl.get("width", 0)
|
|
409
|
+
height = dl.get("height", 0)
|
|
410
|
+
return YtVideoInfo(
|
|
411
|
+
title=title,
|
|
412
|
+
description=description,
|
|
413
|
+
thumbnail=thumbnail,
|
|
414
|
+
duration=duration,
|
|
415
|
+
url=url,
|
|
416
|
+
width=width,
|
|
417
|
+
height=height,
|
|
418
|
+
info_json=dl,
|
|
419
|
+
)
|
|
420
|
+
|
|
421
|
+
async def _extract_info(self, url: str) -> dict[str, Any]:
|
|
422
|
+
return await _run_ytdlp_json(
|
|
423
|
+
url,
|
|
424
|
+
self.cli_args,
|
|
425
|
+
proxy=self.proxy,
|
|
426
|
+
cookie_text=self.get_cookie_text(),
|
|
427
|
+
)
|
|
428
|
+
|
|
429
|
+
def get_cookie_text(self) -> str | None:
|
|
430
|
+
return None
|
|
431
|
+
|
|
432
|
+
@property
|
|
433
|
+
def cli_args(self) -> list[str]:
|
|
434
|
+
return [
|
|
435
|
+
"--quiet", # 不输出日志
|
|
436
|
+
"--no-progress", # 不输出下载进度
|
|
437
|
+
"--no-playlist",
|
|
438
|
+
"--dump-single-json",
|
|
439
|
+
"--no-download",
|
|
440
|
+
"--no-warnings",
|
|
441
|
+
]
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
class YtVideoParseResult(VideoParseResult):
|
|
445
|
+
def __init__(
|
|
446
|
+
self,
|
|
447
|
+
dl: "YtVideoInfo",
|
|
448
|
+
title: str | None,
|
|
449
|
+
video: VideoRef | None = None,
|
|
450
|
+
content: str | None = None,
|
|
451
|
+
):
|
|
452
|
+
"""dl: yt-dlp解析结果"""
|
|
453
|
+
self.dl = dl
|
|
454
|
+
super().__init__(title=title, video=video, content=content)
|
|
455
|
+
|
|
456
|
+
@property
|
|
457
|
+
def cli_args(self) -> list[str]:
|
|
458
|
+
return [
|
|
459
|
+
"--quiet", # 不输出日志
|
|
460
|
+
"--no-progress", # 不输出下载进度
|
|
461
|
+
]
|
|
462
|
+
|
|
463
|
+
async def _do_download(
|
|
464
|
+
self,
|
|
465
|
+
*,
|
|
466
|
+
output_dir: Path,
|
|
467
|
+
callback: ProgressCallback | None = None,
|
|
468
|
+
callback_args: tuple = (),
|
|
469
|
+
callback_kwargs: dict | None = None,
|
|
470
|
+
proxy: str | None = None,
|
|
471
|
+
headers: dict | None = None,
|
|
472
|
+
connections: int = 4,
|
|
473
|
+
) -> "DownloadResult":
|
|
474
|
+
if callback_kwargs is None:
|
|
475
|
+
callback_kwargs = {}
|
|
476
|
+
output_dir_path = Path(output_dir)
|
|
477
|
+
|
|
478
|
+
cli_args = self.cli_args.copy()
|
|
479
|
+
outtmpl = f"{output_dir_path.joinpath(self.name)}.%(ext)s"
|
|
480
|
+
|
|
481
|
+
await self._run_download(
|
|
482
|
+
cli_args,
|
|
483
|
+
outtmpl=outtmpl,
|
|
484
|
+
connections=connections,
|
|
485
|
+
proxy=proxy,
|
|
486
|
+
headers=headers,
|
|
487
|
+
callback=callback,
|
|
488
|
+
callback_args=callback_args,
|
|
489
|
+
callback_kwargs=callback_kwargs,
|
|
490
|
+
)
|
|
491
|
+
|
|
492
|
+
v = [p for p in output_dir_path.glob(f"{self.name}.*") if p.is_file()]
|
|
493
|
+
if not v:
|
|
494
|
+
raise DownloadError("下载失败: 未找到下载后的视频文件")
|
|
495
|
+
|
|
496
|
+
if callback:
|
|
497
|
+
await callback(100, 100, "bytes", *callback_args, **callback_kwargs)
|
|
498
|
+
|
|
499
|
+
video_path = v[0]
|
|
500
|
+
return DownloadResult(
|
|
501
|
+
VideoFile(
|
|
502
|
+
path=str(video_path),
|
|
503
|
+
height=self.dl.height,
|
|
504
|
+
width=self.dl.width,
|
|
505
|
+
duration=self.dl.duration,
|
|
506
|
+
),
|
|
507
|
+
output_dir,
|
|
508
|
+
)
|
|
509
|
+
|
|
510
|
+
async def _run_download(
|
|
511
|
+
self,
|
|
512
|
+
cli_args: list[str],
|
|
513
|
+
*,
|
|
514
|
+
outtmpl: str,
|
|
515
|
+
connections: int,
|
|
516
|
+
proxy: str | None = None,
|
|
517
|
+
headers: dict | None = None,
|
|
518
|
+
callback: ProgressCallback | None = None,
|
|
519
|
+
callback_args: tuple = (),
|
|
520
|
+
callback_kwargs: dict | None = None,
|
|
521
|
+
) -> None:
|
|
522
|
+
try:
|
|
523
|
+
await _run_ytdlp_download(
|
|
524
|
+
self.dl.info_json,
|
|
525
|
+
cli_args,
|
|
526
|
+
outtmpl=outtmpl,
|
|
527
|
+
connections=connections,
|
|
528
|
+
proxy=proxy,
|
|
529
|
+
headers=headers,
|
|
530
|
+
callback=callback,
|
|
531
|
+
callback_args=callback_args,
|
|
532
|
+
callback_kwargs=callback_kwargs,
|
|
533
|
+
)
|
|
534
|
+
except Exception as e:
|
|
535
|
+
raise DownloadError(f"下载失败: {str(e)}") from e
|
|
536
|
+
|
|
537
|
+
|
|
538
|
+
@dataclass
|
|
539
|
+
class YtVideoInfo:
|
|
540
|
+
"""raw_video_info: yt-dlp解析结果"""
|
|
541
|
+
|
|
542
|
+
title: str
|
|
543
|
+
description: str
|
|
544
|
+
thumbnail: str
|
|
545
|
+
url: str
|
|
546
|
+
info_json: dict[str, Any]
|
|
547
|
+
duration: int = 0
|
|
548
|
+
width: int = 0
|
|
549
|
+
height: int = 0
|