parsehub 2.1.4__tar.gz → 2.1.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {parsehub-2.1.4/src/parsehub.egg-info → parsehub-2.1.6}/PKG-INFO +4 -1
- {parsehub-2.1.4 → parsehub-2.1.6}/README.md +3 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/pyproject.toml +1 -1
- parsehub-2.1.6/src/parsehub/parsers/parser/threads.py +39 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/xhs.py +2 -2
- parsehub-2.1.6/src/parsehub/provider_api/threads.py +287 -0
- {parsehub-2.1.4 → parsehub-2.1.6/src/parsehub.egg-info}/PKG-INFO +4 -1
- parsehub-2.1.4/src/parsehub/parsers/parser/threads.py +0 -25
- parsehub-2.1.4/src/parsehub/provider_api/threads.py +0 -174
- {parsehub-2.1.4 → parsehub-2.1.6}/LICENSE +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/setup.cfg +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/__init__.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/__init__.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/cli.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/cli_config.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/config/__init__.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/config/config.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/errors.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/__init__.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/base/__init__.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/base/base.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/base/ytdlp.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/__init__.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/bilibili.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/coolapk.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/douyin.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/facebook.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/instagram.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/kuaishou.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/pipix.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/snapchat.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/tieba.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/tiktok.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/twitter.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/weibo.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/weixin.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/xiaoheihe.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/youtube.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/zhihu.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/parsers/parser/zuiyou.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/provider_api/__init__.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/provider_api/bilibili.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/provider_api/coolapk.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/provider_api/douyin.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/provider_api/instagram.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/provider_api/kuaishou.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/provider_api/pipix.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/provider_api/tieba.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/provider_api/tiktok.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/provider_api/twitter.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/provider_api/weibo.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/provider_api/weixin.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/provider_api/xhs.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/provider_api/xiaoheihe.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/provider_api/zhihu.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/provider_api/zuiyou.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/types/__init__.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/types/callback.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/types/media_file.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/types/media_ref.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/types/platform.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/types/post.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/types/result.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/utils/downloader.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/utils/helpers.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub/utils/media_info.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub.egg-info/SOURCES.txt +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub.egg-info/dependency_links.txt +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub.egg-info/entry_points.txt +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub.egg-info/requires.txt +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/src/parsehub.egg-info/top_level.txt +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/test/test_cli.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/test/test_cli_config.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/test/test_core_offline.py +0 -0
- {parsehub-2.1.4 → parsehub-2.1.6}/test/test_downloader.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: parsehub
|
|
3
|
-
Version: 2.1.
|
|
3
|
+
Version: 2.1.6
|
|
4
4
|
Summary: 轻量、异步、开箱即用的社交媒体聚合解析库
|
|
5
5
|
Author-email: 梓澪 <zilingmio@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -53,6 +53,8 @@ Dynamic: license-file
|
|
|
53
53
|
|
|
54
54
|
轻量, 异步, 开箱即用的社交媒体解析与媒体下载库, 支持 17+ 平台
|
|
55
55
|
|
|
56
|
+
简体中文 | [English](README.en.md)
|
|
57
|
+
|
|
56
58
|
[安装](#-安装) · [快速开始](#-快速开始) · [高级用法](#-高级用法) · [TG Bot](https://github.com/z-mio/parse_hub_bot)
|
|
57
59
|
|
|
58
60
|
</div>
|
|
@@ -205,6 +207,7 @@ print(result)
|
|
|
205
207
|
|
|
206
208
|
- `Twitter / X`
|
|
207
209
|
- `Instagram`
|
|
210
|
+
- `Threads`
|
|
208
211
|
- `YouTube`
|
|
209
212
|
- `Bilibili`
|
|
210
213
|
- `抖音`
|
|
@@ -11,6 +11,8 @@
|
|
|
11
11
|
|
|
12
12
|
轻量, 异步, 开箱即用的社交媒体解析与媒体下载库, 支持 17+ 平台
|
|
13
13
|
|
|
14
|
+
简体中文 | [English](README.en.md)
|
|
15
|
+
|
|
14
16
|
[安装](#-安装) · [快速开始](#-快速开始) · [高级用法](#-高级用法) · [TG Bot](https://github.com/z-mio/parse_hub_bot)
|
|
15
17
|
|
|
16
18
|
</div>
|
|
@@ -163,6 +165,7 @@ print(result)
|
|
|
163
165
|
|
|
164
166
|
- `Twitter / X`
|
|
165
167
|
- `Instagram`
|
|
168
|
+
- `Threads`
|
|
166
169
|
- `YouTube`
|
|
167
170
|
- `Bilibili`
|
|
168
171
|
- `抖音`
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
from ...provider_api.threads import ThreadsAPI, ThreadsAPIError, ThreadsMedia, ThreadsMediaType, ThreadsPost
|
|
2
|
+
from ...types import AnyMediaRef, ImageRef, MultimediaParseResult, ParseError, Platform, VideoRef
|
|
3
|
+
from ..base.base import BaseParser
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class ThreadsParser(BaseParser):
|
|
7
|
+
__platform__ = Platform.THREADS
|
|
8
|
+
__supported_type__ = ["视频", "图文"]
|
|
9
|
+
__match__ = r"^(http(s)?://)?.+threads.com/@[\w.]+/post/.*"
|
|
10
|
+
|
|
11
|
+
async def _do_parse(self, raw_url: str) -> "MultimediaParseResult":
|
|
12
|
+
post = await self._parse(raw_url)
|
|
13
|
+
media: list[AnyMediaRef] = []
|
|
14
|
+
if post.media:
|
|
15
|
+
pm: list[ThreadsMedia] = post.media if isinstance(post.media, list) else [post.media]
|
|
16
|
+
for m in pm:
|
|
17
|
+
match m.type:
|
|
18
|
+
case ThreadsMediaType.VIDEO:
|
|
19
|
+
media.append(VideoRef(url=m.url, thumb_url=m.thumb_url, width=m.width, height=m.height))
|
|
20
|
+
case ThreadsMediaType.IMAGE:
|
|
21
|
+
media.append(ImageRef(url=m.url, thumb_url=m.url, width=m.width, height=m.height))
|
|
22
|
+
return MultimediaParseResult(content=post.content, media=media)
|
|
23
|
+
|
|
24
|
+
async def _parse(self, url: str) -> ThreadsPost:
|
|
25
|
+
# 公开帖子无需登录即可解析; 登录墙内容 (私密/受限/年龄限制) 才需要 Cookie, 有则带上
|
|
26
|
+
try:
|
|
27
|
+
api = ThreadsAPI(proxy=self.proxy, cookie=self.cookie.get_value() if self.cookie else None)
|
|
28
|
+
return await api.parse(url)
|
|
29
|
+
except ThreadsAPIError as e:
|
|
30
|
+
if not self.cookie:
|
|
31
|
+
raise ParseError("无法获取帖子内容: 该帖子可能位于登录墙内, 请为 threads 平台配置 Cookie") from e
|
|
32
|
+
raise ParseError("无法获取帖子内容(可能为私人或受限内容, 或 Cookie 已失效)") from e
|
|
33
|
+
except ParseError:
|
|
34
|
+
raise
|
|
35
|
+
except Exception as e:
|
|
36
|
+
raise ParseError(f"无法获取帖子内容: {e}") from e
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
__all__ = ["ThreadsParser"]
|
|
@@ -20,8 +20,8 @@ from ..base import BaseParser
|
|
|
20
20
|
class XHSParser(BaseParser):
|
|
21
21
|
__platform__ = Platform.XHS
|
|
22
22
|
__supported_type__ = ["视频", "图文"]
|
|
23
|
-
__match__ = r"^(http(s)?://)?.+(xiaohongshu|xhslink).com/.+"
|
|
24
|
-
__redirect_keywords__ = ["xhslink"
|
|
23
|
+
__match__ = r"^(http(s)?://)?.+(xiaohongshu|xhslink).(com|cn)/.+"
|
|
24
|
+
__redirect_keywords__ = ["xhslink"]
|
|
25
25
|
__after_clean_parameters__ = ["xsec_token"]
|
|
26
26
|
|
|
27
27
|
async def _do_parse(self, raw_url: str) -> Union["VideoParseResult", "ImageParseResult", "MultimediaParseResult"]:
|
|
@@ -0,0 +1,287 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from enum import Enum
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
import httpx
|
|
10
|
+
|
|
11
|
+
from ..utils.helpers import UA
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class ThreadsAPIError(Exception):
|
|
15
|
+
"""Threads API 相关错误"""
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class ThreadsAPI:
|
|
19
|
+
GRAPHQL_URL = "https://www.threads.com/graphql/query"
|
|
20
|
+
THREADS_URL = "https://www.threads.com/"
|
|
21
|
+
# BarcelonaPostPageDirectQuery, 通过帖子 ID 获取帖子内容
|
|
22
|
+
POST_DOC_ID = "27419285281047858"
|
|
23
|
+
X_IG_APP_ID = "238260118697367"
|
|
24
|
+
# shortcode <-> pk 使用与 Instagram 相同的 base64 字母表
|
|
25
|
+
SHORTCODE_ALPHABET = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789-_"
|
|
26
|
+
|
|
27
|
+
# Threads 与 Instagram 共用 Meta 的登录后端, 因此使用同一套 Cookie 键
|
|
28
|
+
DEFAULT_COOKIES = {
|
|
29
|
+
"sessionid": "",
|
|
30
|
+
"ds_user_id": "",
|
|
31
|
+
"csrftoken": "",
|
|
32
|
+
"mid": "",
|
|
33
|
+
"ig_did": "",
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
# GraphQL 查询强制要求的 relay provider 变量: 必须全部传入 (值统一给 False 即可),
|
|
37
|
+
# 缺失或只传部分都会导致查询返回 execution error.
|
|
38
|
+
# 与 Instagram 的 doc_id 一样, 该列表会随 Threads 前端更新而变化, 需要时同步维护.
|
|
39
|
+
RELAY_PROVIDERS = (
|
|
40
|
+
"BarcelonaHasPermalinkIndentation",
|
|
41
|
+
"BarcelonaIsLoggedIn",
|
|
42
|
+
"BarcelonaHasPostAuthorNotifControls",
|
|
43
|
+
"BarcelonaShouldShowFediverseM1Features",
|
|
44
|
+
"BarcelonaHasPermalinkPodcastCard",
|
|
45
|
+
"BarcelonaHasDearAlgoConsumption",
|
|
46
|
+
"BarcelonaHasEventBadge",
|
|
47
|
+
"BarcelonaGenAIRepliesEnabled",
|
|
48
|
+
"BarcelonaIsSearchDiscoveryEnabled",
|
|
49
|
+
"BarcelonaHasCommunities",
|
|
50
|
+
"BarcelonaHasGameScoreShare",
|
|
51
|
+
"BarcelonaHasPublicViewCountCard",
|
|
52
|
+
"BarcelonaHasCommunityEntityCard",
|
|
53
|
+
"BarcelonaHasScorecardCommunity",
|
|
54
|
+
"BarcelonaHasSportTeamAllegianceCard",
|
|
55
|
+
"BarcelonaHasMusic",
|
|
56
|
+
"BarcelonaHasNewspaperLinkStyle",
|
|
57
|
+
"BarcelonaHasMessaging",
|
|
58
|
+
"BarcelonaHasPodcastTextFragments",
|
|
59
|
+
"BarcelonaShouldFulfillLightboxQuery",
|
|
60
|
+
"BarcelonaHasViewerReplied",
|
|
61
|
+
"BarcelonaHasPrivateRepliesDeprecation",
|
|
62
|
+
"BarcelonaHasGhostPostEmojiActivation",
|
|
63
|
+
"BarcelonaOptionalCookiesEnabled",
|
|
64
|
+
"BarcelonaHasDearAlgoWebProduction",
|
|
65
|
+
"BarcelonaHasWebFavicons",
|
|
66
|
+
"BarcelonaIsCrawler",
|
|
67
|
+
"BarcelonaHasCommunityTopContributors",
|
|
68
|
+
"BarcelonaCanSeeSponsoredContent",
|
|
69
|
+
"BarcelonaShouldShowFediverseM075Features",
|
|
70
|
+
"BarcelonaIsInternalUser",
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
def __init__(
|
|
74
|
+
self,
|
|
75
|
+
proxy: str | None = None,
|
|
76
|
+
cookie: dict[str, str] | None = None,
|
|
77
|
+
timeout: float = 30,
|
|
78
|
+
):
|
|
79
|
+
self.proxy = proxy
|
|
80
|
+
self.cookie = cookie or {}
|
|
81
|
+
self.timeout = timeout
|
|
82
|
+
|
|
83
|
+
async def parse(self, url: str) -> ThreadsPost:
|
|
84
|
+
code = self.get_post_id_by_url(url)
|
|
85
|
+
payload = await self._post_graphql(
|
|
86
|
+
doc_id=self.POST_DOC_ID,
|
|
87
|
+
variables=self._build_variables(code),
|
|
88
|
+
)
|
|
89
|
+
post = self._extract_post(payload, code)
|
|
90
|
+
if post is None:
|
|
91
|
+
raise ThreadsAPIError("Fetching Post metadata failed.")
|
|
92
|
+
return ThreadsPost.from_graphql(post)
|
|
93
|
+
|
|
94
|
+
def _build_variables(self, code: str) -> dict[str, Any]:
|
|
95
|
+
variables: dict[str, Any] = {"postID": str(self.shortcode_to_pk(code))}
|
|
96
|
+
for name in self.RELAY_PROVIDERS:
|
|
97
|
+
variables[f"__relay_internal__pv__{name}relayprovider"] = False
|
|
98
|
+
return variables
|
|
99
|
+
|
|
100
|
+
async def _post_graphql(self, *, doc_id: str, variables: dict[str, Any]) -> dict[str, Any]:
|
|
101
|
+
async with self._new_client() as client:
|
|
102
|
+
await self._ensure_csrf_token(client)
|
|
103
|
+
data = {
|
|
104
|
+
"variables": json.dumps(variables, separators=(",", ":")),
|
|
105
|
+
"doc_id": doc_id,
|
|
106
|
+
"server_timestamps": "true",
|
|
107
|
+
}
|
|
108
|
+
try:
|
|
109
|
+
response = await client.post(self.GRAPHQL_URL, data=data, follow_redirects=False)
|
|
110
|
+
except httpx.HTTPError as exc:
|
|
111
|
+
raise ThreadsAPIError(f"请求 Threads GraphQL 失败: {exc}") from exc
|
|
112
|
+
|
|
113
|
+
if response.status_code != 200:
|
|
114
|
+
raise ThreadsAPIError(f"Threads GraphQL 返回 HTTP {response.status_code}: {response.text[:500]}")
|
|
115
|
+
|
|
116
|
+
try:
|
|
117
|
+
payload: dict[str, Any] = response.json()
|
|
118
|
+
except ValueError as exc:
|
|
119
|
+
# 未登录时会返回 HTML 登录页
|
|
120
|
+
raise ThreadsAPIError("Threads GraphQL 返回非 JSON 响应(可能需要登录)") from exc
|
|
121
|
+
|
|
122
|
+
if payload.get("errors"):
|
|
123
|
+
raise ThreadsAPIError(f"Threads GraphQL 返回错误: {payload['errors']}")
|
|
124
|
+
return payload
|
|
125
|
+
|
|
126
|
+
@staticmethod
|
|
127
|
+
def _extract_post(payload: dict[str, Any], code: str) -> dict[str, Any] | None:
|
|
128
|
+
data = ((payload.get("data") or {}).get("data")) or {}
|
|
129
|
+
edges = data.get("edges") or []
|
|
130
|
+
fallback: dict[str, Any] | None = None
|
|
131
|
+
for edge in edges:
|
|
132
|
+
for item in (edge.get("node") or {}).get("thread_items") or []:
|
|
133
|
+
post = item.get("post")
|
|
134
|
+
if not isinstance(post, dict):
|
|
135
|
+
continue
|
|
136
|
+
if fallback is None:
|
|
137
|
+
fallback = post
|
|
138
|
+
if post.get("code") == code:
|
|
139
|
+
return post
|
|
140
|
+
# 找不到精确匹配时退回第一条 (通常即目标帖子本身)
|
|
141
|
+
return fallback
|
|
142
|
+
|
|
143
|
+
def _new_client(self) -> httpx.AsyncClient:
|
|
144
|
+
cookies = self.DEFAULT_COOKIES | self.cookie
|
|
145
|
+
headers = {
|
|
146
|
+
"Accept": "*/*",
|
|
147
|
+
"Content-Type": "application/x-www-form-urlencoded",
|
|
148
|
+
"Referer": self.THREADS_URL,
|
|
149
|
+
"User-Agent": UA,
|
|
150
|
+
"X-IG-App-ID": self.X_IG_APP_ID,
|
|
151
|
+
}
|
|
152
|
+
return httpx.AsyncClient(
|
|
153
|
+
cookies=cookies,
|
|
154
|
+
headers=headers,
|
|
155
|
+
proxy=self.proxy,
|
|
156
|
+
timeout=self.timeout,
|
|
157
|
+
)
|
|
158
|
+
|
|
159
|
+
async def _ensure_csrf_token(self, client: httpx.AsyncClient) -> None:
|
|
160
|
+
csrf_token = self._get_cookie_value(client, "csrftoken")
|
|
161
|
+
if not csrf_token:
|
|
162
|
+
try:
|
|
163
|
+
await client.get(self.THREADS_URL, follow_redirects=True)
|
|
164
|
+
except httpx.HTTPError as exc:
|
|
165
|
+
raise ThreadsAPIError(f"获取 Threads csrftoken 失败: {exc}") from exc
|
|
166
|
+
csrf_token = self._get_cookie_value(client, "csrftoken")
|
|
167
|
+
if csrf_token:
|
|
168
|
+
client.headers["x-csrftoken"] = csrf_token
|
|
169
|
+
|
|
170
|
+
@staticmethod
|
|
171
|
+
def _get_cookie_value(client: httpx.AsyncClient, name: str) -> str:
|
|
172
|
+
values = [cookie.value for cookie in client.cookies.jar if cookie.name == name and cookie.value]
|
|
173
|
+
return values[-1] if values else ""
|
|
174
|
+
|
|
175
|
+
@classmethod
|
|
176
|
+
def shortcode_to_pk(cls, code: str) -> int:
|
|
177
|
+
pk = 0
|
|
178
|
+
for ch in code:
|
|
179
|
+
try:
|
|
180
|
+
pk = pk * 64 + cls.SHORTCODE_ALPHABET.index(ch)
|
|
181
|
+
except ValueError as exc:
|
|
182
|
+
raise ValueError(f"无效的 Threads 帖子 ID: {code}") from exc
|
|
183
|
+
return pk
|
|
184
|
+
|
|
185
|
+
@staticmethod
|
|
186
|
+
def get_username_by_url(url: str) -> str:
|
|
187
|
+
u = re.search(r"/(@[\w.]+)/post/", url)
|
|
188
|
+
if not u:
|
|
189
|
+
raise ValueError("从 URL 中获取用户名失败")
|
|
190
|
+
return u[1]
|
|
191
|
+
|
|
192
|
+
@staticmethod
|
|
193
|
+
def get_post_id_by_url(url: str) -> str:
|
|
194
|
+
p = re.search(r"/post/([\w-]+)", url)
|
|
195
|
+
if not p:
|
|
196
|
+
raise ValueError("从 URL 中获取帖子 ID 失败")
|
|
197
|
+
return p[1]
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
class ThreadsMediaType(Enum):
|
|
201
|
+
IMAGE = "image"
|
|
202
|
+
VIDEO = "video"
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
@dataclass
|
|
206
|
+
class ThreadsMedia:
|
|
207
|
+
type: ThreadsMediaType
|
|
208
|
+
url: str
|
|
209
|
+
thumb_url: str | None = None
|
|
210
|
+
width: int = 0
|
|
211
|
+
height: int = 0
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
@dataclass
|
|
215
|
+
class ThreadsPost:
|
|
216
|
+
content: str
|
|
217
|
+
media: ThreadsMedia | list[ThreadsMedia] | None = None
|
|
218
|
+
|
|
219
|
+
@classmethod
|
|
220
|
+
def from_graphql(cls, post: dict[str, Any]) -> ThreadsPost:
|
|
221
|
+
caption = post.get("caption")
|
|
222
|
+
content = caption.get("text") if isinstance(caption, dict) else caption
|
|
223
|
+
return cls(content=str(content or ""), media=cls._fetch_media(post))
|
|
224
|
+
|
|
225
|
+
@classmethod
|
|
226
|
+
def _fetch_media(cls, d: dict[str, Any]) -> ThreadsMedia | list[ThreadsMedia]:
|
|
227
|
+
media: ThreadsMedia | list[ThreadsMedia]
|
|
228
|
+
match d.get("media_type"):
|
|
229
|
+
case 1: # 单张图片
|
|
230
|
+
image = d["image_versions2"]["candidates"][0]
|
|
231
|
+
media = ThreadsMedia(
|
|
232
|
+
type=ThreadsMediaType.IMAGE,
|
|
233
|
+
url=image["url"],
|
|
234
|
+
thumb_url=image["url"],
|
|
235
|
+
width=image.get("width", 0),
|
|
236
|
+
height=image.get("height", 0),
|
|
237
|
+
)
|
|
238
|
+
case 2: # 单个视频
|
|
239
|
+
thumb = d["image_versions2"]["candidates"][0]["url"]
|
|
240
|
+
video = d["video_versions"][0]["url"]
|
|
241
|
+
media = ThreadsMedia(
|
|
242
|
+
type=ThreadsMediaType.VIDEO,
|
|
243
|
+
url=video,
|
|
244
|
+
thumb_url=thumb,
|
|
245
|
+
width=d.get("original_width", 0),
|
|
246
|
+
height=d.get("original_height", 0),
|
|
247
|
+
)
|
|
248
|
+
case 8: # 多图/视频
|
|
249
|
+
media = []
|
|
250
|
+
for m in d.get("carousel_media") or []:
|
|
251
|
+
if m.get("video_versions"):
|
|
252
|
+
thumb = m["image_versions2"]["candidates"][0]["url"]
|
|
253
|
+
media.append(
|
|
254
|
+
ThreadsMedia(
|
|
255
|
+
type=ThreadsMediaType.VIDEO,
|
|
256
|
+
url=m["video_versions"][0]["url"],
|
|
257
|
+
thumb_url=thumb,
|
|
258
|
+
width=m.get("original_width", 0),
|
|
259
|
+
height=m.get("original_height", 0),
|
|
260
|
+
)
|
|
261
|
+
)
|
|
262
|
+
else:
|
|
263
|
+
image = m["image_versions2"]["candidates"][0]
|
|
264
|
+
media.append(
|
|
265
|
+
ThreadsMedia(
|
|
266
|
+
type=ThreadsMediaType.IMAGE,
|
|
267
|
+
url=image["url"],
|
|
268
|
+
thumb_url=image["url"],
|
|
269
|
+
width=m.get("original_width", 0),
|
|
270
|
+
height=m.get("original_height", 0),
|
|
271
|
+
)
|
|
272
|
+
)
|
|
273
|
+
case 19: # 纯文本/外部链接
|
|
274
|
+
linked = (d.get("text_post_app_info") or {}).get("linked_inline_media")
|
|
275
|
+
media = cls._fetch_media(linked) if linked else []
|
|
276
|
+
case _:
|
|
277
|
+
media = []
|
|
278
|
+
return media
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
__all__ = [
|
|
282
|
+
"ThreadsAPI",
|
|
283
|
+
"ThreadsAPIError",
|
|
284
|
+
"ThreadsMedia",
|
|
285
|
+
"ThreadsMediaType",
|
|
286
|
+
"ThreadsPost",
|
|
287
|
+
]
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: parsehub
|
|
3
|
-
Version: 2.1.
|
|
3
|
+
Version: 2.1.6
|
|
4
4
|
Summary: 轻量、异步、开箱即用的社交媒体聚合解析库
|
|
5
5
|
Author-email: 梓澪 <zilingmio@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -53,6 +53,8 @@ Dynamic: license-file
|
|
|
53
53
|
|
|
54
54
|
轻量, 异步, 开箱即用的社交媒体解析与媒体下载库, 支持 17+ 平台
|
|
55
55
|
|
|
56
|
+
简体中文 | [English](README.en.md)
|
|
57
|
+
|
|
56
58
|
[安装](#-安装) · [快速开始](#-快速开始) · [高级用法](#-高级用法) · [TG Bot](https://github.com/z-mio/parse_hub_bot)
|
|
57
59
|
|
|
58
60
|
</div>
|
|
@@ -205,6 +207,7 @@ print(result)
|
|
|
205
207
|
|
|
206
208
|
- `Twitter / X`
|
|
207
209
|
- `Instagram`
|
|
210
|
+
- `Threads`
|
|
208
211
|
- `YouTube`
|
|
209
212
|
- `Bilibili`
|
|
210
213
|
- `抖音`
|
|
@@ -1,25 +0,0 @@
|
|
|
1
|
-
from ...provider_api.threads import ThreadsAPI, ThreadsMedia, ThreadsMediaType
|
|
2
|
-
from ...types import AnyMediaRef, ImageRef, MultimediaParseResult, Platform, VideoRef
|
|
3
|
-
from ..base.base import BaseParser
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
class ThreadsParser(BaseParser):
|
|
7
|
-
__platform__ = Platform.THREADS
|
|
8
|
-
__supported_type__ = ["视频", "图文"]
|
|
9
|
-
__match__ = r"^(http(s)?://)?.+threads.com/@[\w.]+/post/.*"
|
|
10
|
-
|
|
11
|
-
async def _do_parse(self, raw_url: str) -> "MultimediaParseResult":
|
|
12
|
-
post = await ThreadsAPI(proxy=self.proxy).parse(raw_url)
|
|
13
|
-
media: list[AnyMediaRef] = []
|
|
14
|
-
if post.media:
|
|
15
|
-
pm: list[ThreadsMedia] = post.media if isinstance(post.media, list) else [post.media]
|
|
16
|
-
for m in pm:
|
|
17
|
-
match m.type:
|
|
18
|
-
case ThreadsMediaType.VIDEO:
|
|
19
|
-
media.append(VideoRef(url=m.url, thumb_url=m.thumb_url, width=m.width, height=m.height))
|
|
20
|
-
case ThreadsMediaType.IMAGE:
|
|
21
|
-
media.append(ImageRef(url=m.url, thumb_url=m.url, width=m.width, height=m.height))
|
|
22
|
-
return MultimediaParseResult(content=post.content, media=media)
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
__all__ = ["ThreadsParser"]
|
|
@@ -1,174 +0,0 @@
|
|
|
1
|
-
from __future__ import annotations
|
|
2
|
-
|
|
3
|
-
import json
|
|
4
|
-
import random
|
|
5
|
-
import re
|
|
6
|
-
import string
|
|
7
|
-
from dataclasses import dataclass
|
|
8
|
-
from enum import Enum
|
|
9
|
-
|
|
10
|
-
import httpx
|
|
11
|
-
|
|
12
|
-
from ..utils.helpers import UA
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
class ThreadsAPI:
|
|
16
|
-
def __init__(self, proxy: str | None = None):
|
|
17
|
-
self.proxy = proxy
|
|
18
|
-
|
|
19
|
-
async def parse(self, url: str) -> ThreadsPost:
|
|
20
|
-
lsd = self.random_lsd()
|
|
21
|
-
headers = {
|
|
22
|
-
"content-type": "application/x-www-form-urlencoded",
|
|
23
|
-
"sec-fetch-site": "same-origin",
|
|
24
|
-
"user-agent": UA,
|
|
25
|
-
"x-fb-lsd": lsd,
|
|
26
|
-
}
|
|
27
|
-
|
|
28
|
-
data = {
|
|
29
|
-
"route_url": f"/{self.get_username_by_url(url)}/post/{self.get_post_id_by_url(url)}/media",
|
|
30
|
-
"routing_namespace": "barcelona_web",
|
|
31
|
-
"__user": "0",
|
|
32
|
-
"__a": "1",
|
|
33
|
-
"__req": "m",
|
|
34
|
-
"__comet_req": "29",
|
|
35
|
-
"lsd": lsd,
|
|
36
|
-
}
|
|
37
|
-
|
|
38
|
-
async with httpx.AsyncClient(proxy=self.proxy) as client:
|
|
39
|
-
response = await client.post("https://www.threads.com/ajax/route-definition", headers=headers, data=data)
|
|
40
|
-
response.raise_for_status()
|
|
41
|
-
jsonp = [json.loads(j.strip()) for j in response.text.strip().split("for (;;);") if j]
|
|
42
|
-
return ThreadsPost.parse(jsonp)
|
|
43
|
-
|
|
44
|
-
@staticmethod
|
|
45
|
-
def get_username_by_url(url: str) -> str:
|
|
46
|
-
u = re.search(r"/(@[\w.]+)/post/", url)
|
|
47
|
-
if not u:
|
|
48
|
-
raise ValueError("从 URL 中获取用户名失败")
|
|
49
|
-
return u[1]
|
|
50
|
-
|
|
51
|
-
@staticmethod
|
|
52
|
-
def get_post_id_by_url(url: str) -> str:
|
|
53
|
-
p = re.search(r"/post/([\w-]+)", url)
|
|
54
|
-
if not p:
|
|
55
|
-
raise ValueError("从 URL 中获取帖子 ID 失败")
|
|
56
|
-
return p[1]
|
|
57
|
-
|
|
58
|
-
@staticmethod
|
|
59
|
-
def random_lsd() -> str:
|
|
60
|
-
return "".join(random.sample(string.ascii_letters + string.digits, 11))
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
class ThreadsMediaType(Enum):
|
|
64
|
-
IMAGE = "image"
|
|
65
|
-
VIDEO = "video"
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
@dataclass
|
|
69
|
-
class ThreadsMedia:
|
|
70
|
-
type: ThreadsMediaType
|
|
71
|
-
url: str
|
|
72
|
-
thumb_url: str | None = None
|
|
73
|
-
width: int = 0
|
|
74
|
-
height: int = 0
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
@dataclass
|
|
78
|
-
class ThreadsPost:
|
|
79
|
-
content: str
|
|
80
|
-
media: ThreadsMedia | list[ThreadsMedia] | None = None
|
|
81
|
-
|
|
82
|
-
@classmethod
|
|
83
|
-
def parse(cls, jsonp: list[dict]) -> ThreadsPost:
|
|
84
|
-
content = ""
|
|
85
|
-
media: ThreadsMedia | list[ThreadsMedia] | None = []
|
|
86
|
-
for j in jsonp:
|
|
87
|
-
match j["__type"]:
|
|
88
|
-
case "first_response":
|
|
89
|
-
content = cls._fetch_content(j)
|
|
90
|
-
case "preloader":
|
|
91
|
-
if "BarcelonaLightboxDialogRootQueryRelayPreloader" in (j.get("id") or ""):
|
|
92
|
-
media = cls._fetch_media(j)
|
|
93
|
-
case "last_response":
|
|
94
|
-
...
|
|
95
|
-
return cls(content=content, media=media)
|
|
96
|
-
|
|
97
|
-
@staticmethod
|
|
98
|
-
def _fetch_content(data: dict) -> str:
|
|
99
|
-
payload = data["payload"]
|
|
100
|
-
result = payload.get("result", {})
|
|
101
|
-
# 尝试从 redirect_result 获取(用户更改名称后的情况)
|
|
102
|
-
meta = result.get("redirect_result", {}).get("exports", {}).get("meta")
|
|
103
|
-
# 如果没有 redirect_result,则从正常路径获取
|
|
104
|
-
if not meta:
|
|
105
|
-
meta = result.get("exports", {}).get("meta")
|
|
106
|
-
if not meta:
|
|
107
|
-
raise Exception("获取内容失败")
|
|
108
|
-
return str(meta["title"])
|
|
109
|
-
|
|
110
|
-
@staticmethod
|
|
111
|
-
def _fetch_media(data: dict) -> ThreadsMedia | list[ThreadsMedia]:
|
|
112
|
-
data = data.get("result", {}).get("result", {}).get("data", {}).get("data")
|
|
113
|
-
if not data:
|
|
114
|
-
return []
|
|
115
|
-
|
|
116
|
-
def fn(d: dict) -> ThreadsMedia | list[ThreadsMedia]:
|
|
117
|
-
media: ThreadsMedia | list[ThreadsMedia]
|
|
118
|
-
match d["media_type"]:
|
|
119
|
-
case 1: # 单张图片
|
|
120
|
-
image = d["image_versions2"]["candidates"][0]
|
|
121
|
-
media = ThreadsMedia(
|
|
122
|
-
type=ThreadsMediaType.IMAGE,
|
|
123
|
-
url=image["url"],
|
|
124
|
-
thumb_url=image["url"],
|
|
125
|
-
width=image["width"],
|
|
126
|
-
height=image["height"],
|
|
127
|
-
)
|
|
128
|
-
case 2: # 单个视频
|
|
129
|
-
thumb = d["image_versions2"]["candidates"][0]["url"]
|
|
130
|
-
video = d["video_versions"][0]["url"]
|
|
131
|
-
media = ThreadsMedia(
|
|
132
|
-
type=ThreadsMediaType.VIDEO,
|
|
133
|
-
url=video,
|
|
134
|
-
thumb_url=thumb,
|
|
135
|
-
width=d["original_width"],
|
|
136
|
-
height=d["original_height"],
|
|
137
|
-
)
|
|
138
|
-
case 8: # 多图/视频
|
|
139
|
-
carousel_media = d["carousel_media"]
|
|
140
|
-
media = []
|
|
141
|
-
for m in carousel_media:
|
|
142
|
-
if m["video_versions"]:
|
|
143
|
-
thumb = m["image_versions2"]["candidates"][0]["url"]
|
|
144
|
-
video = m["video_versions"][0]["url"]
|
|
145
|
-
media.append(
|
|
146
|
-
ThreadsMedia(
|
|
147
|
-
type=ThreadsMediaType.VIDEO,
|
|
148
|
-
url=video,
|
|
149
|
-
thumb_url=thumb,
|
|
150
|
-
width=m["original_width"],
|
|
151
|
-
height=m["original_height"],
|
|
152
|
-
)
|
|
153
|
-
)
|
|
154
|
-
else:
|
|
155
|
-
image = m["image_versions2"]["candidates"][0]["url"]
|
|
156
|
-
media.append(
|
|
157
|
-
ThreadsMedia(
|
|
158
|
-
type=ThreadsMediaType.IMAGE,
|
|
159
|
-
url=image,
|
|
160
|
-
thumb_url=image,
|
|
161
|
-
width=m["original_width"],
|
|
162
|
-
height=m["original_height"],
|
|
163
|
-
)
|
|
164
|
-
)
|
|
165
|
-
case 19: # 纯文本/外部链接
|
|
166
|
-
if linked_inline_media := d["text_post_app_info"]["linked_inline_media"]:
|
|
167
|
-
media = fn(linked_inline_media)
|
|
168
|
-
else:
|
|
169
|
-
media = []
|
|
170
|
-
case _:
|
|
171
|
-
media = []
|
|
172
|
-
return media
|
|
173
|
-
|
|
174
|
-
return fn(data)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|