parsehub 2.0.32__tar.gz → 2.0.34__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {parsehub-2.0.32/src/parsehub.egg-info → parsehub-2.0.34}/PKG-INFO +1 -2
- {parsehub-2.0.32 → parsehub-2.0.34}/pyproject.toml +1 -2
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/instagram.py +9 -27
- parsehub-2.0.34/src/parsehub/provider_api/instagram.py +380 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/xhs.py +17 -2
- {parsehub-2.0.32 → parsehub-2.0.34/src/parsehub.egg-info}/PKG-INFO +1 -2
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub.egg-info/requires.txt +0 -1
- parsehub-2.0.32/src/parsehub/provider_api/instagram.py +0 -79
- {parsehub-2.0.32 → parsehub-2.0.34}/LICENSE +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/README.md +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/setup.cfg +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/__init__.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/__init__.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/cli.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/cli_config.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/config/__init__.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/config/config.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/errors.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/__init__.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/base/__init__.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/base/base.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/base/ytdlp.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/__init__.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/bilibili.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/coolapk.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/douyin.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/facebook.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/kuaishou.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/pipix.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/snapchat.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/threads.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/tieba.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/tiktok.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/twitter.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/weibo.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/weixin.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/xhs.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/xiaoheihe.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/youtube.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/zuiyou.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/__init__.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/bilibili.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/coolapk.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/douyin.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/kuaishou.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/pipix.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/threads.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/tieba.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/tiktok.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/twitter.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/weibo.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/weixin.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/xiaoheihe.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/zuiyou.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/types/__init__.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/types/callback.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/types/media_file.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/types/media_ref.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/types/platform.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/types/post.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/types/result.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/utils/downloader.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/utils/helpers.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/utils/media_info.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub.egg-info/SOURCES.txt +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub.egg-info/dependency_links.txt +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub.egg-info/entry_points.txt +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub.egg-info/top_level.txt +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/test/test_cli.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/test/test_cli_config.py +0 -0
- {parsehub-2.0.32 → parsehub-2.0.34}/test/test_core_offline.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: parsehub
|
|
3
|
-
Version: 2.0.
|
|
3
|
+
Version: 2.0.34
|
|
4
4
|
Summary: 轻量、异步、开箱即用的社交媒体聚合解析库
|
|
5
5
|
Author-email: 梓澪 <zilingmio@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -24,7 +24,6 @@ Requires-Dist: tenacity>=8.5.0
|
|
|
24
24
|
Requires-Dist: urlextract>=1.9.0
|
|
25
25
|
Requires-Dist: yt-dlp[default]
|
|
26
26
|
Requires-Dist: lxml>=5.3.0
|
|
27
|
-
Requires-Dist: instaloader>=4.14
|
|
28
27
|
Requires-Dist: pydantic>=1.10.19
|
|
29
28
|
Requires-Dist: markdownify>=1.1.0
|
|
30
29
|
Requires-Dist: markdown>=3.7
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "parsehub"
|
|
3
|
-
version = "2.0.
|
|
3
|
+
version = "2.0.34"
|
|
4
4
|
description = "轻量、异步、开箱即用的社交媒体聚合解析库"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.12.0"
|
|
@@ -27,7 +27,6 @@ dependencies = [
|
|
|
27
27
|
"urlextract>=1.9.0",
|
|
28
28
|
"yt-dlp[default]",
|
|
29
29
|
"lxml>=5.3.0",
|
|
30
|
-
"instaloader>=4.14",
|
|
31
30
|
"pydantic>=1.10.19",
|
|
32
31
|
"markdownify>=1.1.0",
|
|
33
32
|
"markdown>=3.7",
|
|
@@ -1,10 +1,6 @@
|
|
|
1
|
-
import asyncio
|
|
2
1
|
import re
|
|
3
|
-
from typing import cast
|
|
4
2
|
|
|
5
|
-
from
|
|
6
|
-
|
|
7
|
-
from ...provider_api.instagram import MyInstaloaderContext, MyPost
|
|
3
|
+
from ...provider_api.instagram import InstagramAPI, InstagramAPIError, InstagramMediaType, InstagramPost
|
|
8
4
|
from ...types import ImageParseResult, ImageRef, MultimediaParseResult, ParseError, Platform, VideoParseResult, VideoRef
|
|
9
5
|
from ...utils.helpers import SecretCookie
|
|
10
6
|
from ..base.base import BaseParser
|
|
@@ -23,14 +19,10 @@ class InstagramParser(BaseParser):
|
|
|
23
19
|
|
|
24
20
|
post = await self._parse(raw_url, shortcode)
|
|
25
21
|
|
|
26
|
-
|
|
27
|
-
dimensions: dict = post._field("dimensions")
|
|
28
|
-
except KeyError:
|
|
29
|
-
dimensions = {}
|
|
30
|
-
width, height = dimensions.get("width", 0) or 0, dimensions.get("height", 0) or 0
|
|
22
|
+
width, height = post.width, post.height
|
|
31
23
|
|
|
32
24
|
match post.typename:
|
|
33
|
-
case
|
|
25
|
+
case InstagramMediaType.SIDECAR:
|
|
34
26
|
media = [
|
|
35
27
|
VideoRef(url=i.video_url, thumb_url=i.display_url, width=i.width, height=i.height)
|
|
36
28
|
if i.is_video and i.video_url
|
|
@@ -38,11 +30,11 @@ class InstagramParser(BaseParser):
|
|
|
38
30
|
for i in post.get_sidecar_nodes()
|
|
39
31
|
]
|
|
40
32
|
return MultimediaParseResult(media=media, title=post.title, content=post.caption)
|
|
41
|
-
case
|
|
33
|
+
case InstagramMediaType.IMAGE:
|
|
42
34
|
return ImageParseResult(
|
|
43
35
|
photo=[ImageRef(url=post.url, width=width, height=height)], title=post.title, content=post.caption
|
|
44
36
|
)
|
|
45
|
-
case
|
|
37
|
+
case InstagramMediaType.VIDEO:
|
|
46
38
|
return VideoParseResult(
|
|
47
39
|
video=VideoRef(
|
|
48
40
|
url=post.video_url or post.url,
|
|
@@ -57,19 +49,11 @@ class InstagramParser(BaseParser):
|
|
|
57
49
|
case _:
|
|
58
50
|
raise ParseError("不支持的类型")
|
|
59
51
|
|
|
60
|
-
async def _parse(self, url: str, shortcode: str, cookie: SecretCookie | None = None) ->
|
|
52
|
+
async def _parse(self, url: str, shortcode: str, cookie: SecretCookie | None = None) -> InstagramPost:
|
|
61
53
|
try:
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
MyInstaloaderContext(self.proxy, cookie.get_value() if cookie else None),
|
|
66
|
-
shortcode,
|
|
67
|
-
),
|
|
68
|
-
30,
|
|
69
|
-
)
|
|
70
|
-
except TimeoutError as e:
|
|
71
|
-
raise ParseError("解析超时") from e
|
|
72
|
-
except BadResponseException as e:
|
|
54
|
+
api = InstagramAPI(proxy=self.proxy, cookie=cookie.get_value() if cookie else None, timeout=30)
|
|
55
|
+
return await api.get_post(shortcode)
|
|
56
|
+
except InstagramAPIError as e:
|
|
73
57
|
match str(e):
|
|
74
58
|
case "Fetching Post metadata failed.":
|
|
75
59
|
if self.cookie and cookie is None:
|
|
@@ -84,8 +68,6 @@ class InstagramParser(BaseParser):
|
|
|
84
68
|
else:
|
|
85
69
|
text = str(e)
|
|
86
70
|
raise ParseError(f"无法获取帖子内容: {text}") from e
|
|
87
|
-
else:
|
|
88
|
-
return cast(MyPost, post)
|
|
89
71
|
|
|
90
72
|
@staticmethod
|
|
91
73
|
def get_short_code(url: str) -> str | None:
|
|
@@ -0,0 +1,380 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from collections.abc import Iterator
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from enum import StrEnum
|
|
7
|
+
from typing import Any, cast
|
|
8
|
+
|
|
9
|
+
import httpx
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class InstagramAPIError(RuntimeError):
|
|
13
|
+
"""Instagram 接口请求或响应解析失败。"""
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class InstagramMediaType(StrEnum):
|
|
17
|
+
IMAGE = "GraphImage"
|
|
18
|
+
VIDEO = "GraphVideo"
|
|
19
|
+
SIDECAR = "GraphSidecar"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(slots=True)
|
|
23
|
+
class InstagramSidecarNode:
|
|
24
|
+
is_video: bool
|
|
25
|
+
display_url: str
|
|
26
|
+
video_url: str | None
|
|
27
|
+
width: int
|
|
28
|
+
height: int
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class InstagramPost:
|
|
32
|
+
"""轻量版 Instagram Post,只保留当前项目解析帖子需要的字段。"""
|
|
33
|
+
|
|
34
|
+
_XDT_TYPES = {
|
|
35
|
+
"XDTGraphImage": InstagramMediaType.IMAGE,
|
|
36
|
+
"XDTGraphVideo": InstagramMediaType.VIDEO,
|
|
37
|
+
"XDTGraphSidecar": InstagramMediaType.SIDECAR,
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
def __init__(self, node: dict[str, Any]):
|
|
41
|
+
self._node = node
|
|
42
|
+
self._normalize_typename()
|
|
43
|
+
|
|
44
|
+
def _normalize_typename(self) -> None:
|
|
45
|
+
typename = self._node.get("__typename")
|
|
46
|
+
if typename in self._XDT_TYPES:
|
|
47
|
+
self._node["__typename"] = self._XDT_TYPES[typename]
|
|
48
|
+
|
|
49
|
+
def _field(self, *keys: str) -> Any:
|
|
50
|
+
value: Any = self._node
|
|
51
|
+
for key in keys:
|
|
52
|
+
value = value[key]
|
|
53
|
+
return value
|
|
54
|
+
|
|
55
|
+
@property
|
|
56
|
+
def shortcode(self) -> str:
|
|
57
|
+
return str(self._node.get("shortcode") or self._node["code"])
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def typename(self) -> InstagramMediaType:
|
|
61
|
+
return InstagramMediaType(self._field("__typename"))
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def is_video(self) -> bool:
|
|
65
|
+
return bool(self._field("is_video"))
|
|
66
|
+
|
|
67
|
+
@property
|
|
68
|
+
def title(self) -> str | None:
|
|
69
|
+
return self._node.get("title")
|
|
70
|
+
|
|
71
|
+
@property
|
|
72
|
+
def caption(self) -> str | None:
|
|
73
|
+
caption_edges = self._node.get("edge_media_to_caption", {}).get("edges") or []
|
|
74
|
+
if caption_edges:
|
|
75
|
+
if text := caption_edges[0].get("node", {}).get("text"):
|
|
76
|
+
return str(text)
|
|
77
|
+
return None
|
|
78
|
+
return self._node.get("caption")
|
|
79
|
+
|
|
80
|
+
@property
|
|
81
|
+
def url(self) -> str:
|
|
82
|
+
return str(self._node.get("display_url") or self._node["display_src"])
|
|
83
|
+
|
|
84
|
+
@property
|
|
85
|
+
def video_url(self) -> str | None:
|
|
86
|
+
if not self.is_video:
|
|
87
|
+
return None
|
|
88
|
+
return self._node.get("video_url")
|
|
89
|
+
|
|
90
|
+
@property
|
|
91
|
+
def video_duration(self) -> float | None:
|
|
92
|
+
value = self._node.get("video_duration")
|
|
93
|
+
return float(value) if value is not None else None
|
|
94
|
+
|
|
95
|
+
@property
|
|
96
|
+
def width(self) -> int:
|
|
97
|
+
return int(self._node.get("dimensions", {}).get("width") or 0)
|
|
98
|
+
|
|
99
|
+
@property
|
|
100
|
+
def height(self) -> int:
|
|
101
|
+
return int(self._node.get("dimensions", {}).get("height") or 0)
|
|
102
|
+
|
|
103
|
+
def get_sidecar_nodes(self, start: int = 0, end: int = -1) -> Iterator[InstagramSidecarNode]:
|
|
104
|
+
if self.typename is not InstagramMediaType.SIDECAR:
|
|
105
|
+
return
|
|
106
|
+
|
|
107
|
+
edges = self._field("edge_sidecar_to_children", "edges")
|
|
108
|
+
if end < 0:
|
|
109
|
+
end = len(edges) - 1
|
|
110
|
+
if start < 0:
|
|
111
|
+
start = len(edges) - 1
|
|
112
|
+
|
|
113
|
+
for idx, edge in enumerate(edges):
|
|
114
|
+
if not start <= idx <= end:
|
|
115
|
+
continue
|
|
116
|
+
|
|
117
|
+
node = edge["node"]
|
|
118
|
+
dimensions = node.get("dimensions", {})
|
|
119
|
+
is_video = bool(node.get("is_video"))
|
|
120
|
+
yield InstagramSidecarNode(
|
|
121
|
+
is_video=is_video,
|
|
122
|
+
display_url=node.get("display_url") or node.get("display_src") or "",
|
|
123
|
+
video_url=node.get("video_url") if is_video else None,
|
|
124
|
+
width=int(dimensions.get("width") or 0),
|
|
125
|
+
height=int(dimensions.get("height") or 0),
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
class InstagramAPI:
|
|
130
|
+
GRAPHQL_URL = "https://www.instagram.com/graphql/query"
|
|
131
|
+
INSTAGRAM_URL = "https://www.instagram.com/"
|
|
132
|
+
SHORTCODE_DOC_ID = "27128499623469141"
|
|
133
|
+
|
|
134
|
+
DEFAULT_COOKIES = {
|
|
135
|
+
"sessionid": "",
|
|
136
|
+
"mid": "",
|
|
137
|
+
"ig_pr": "1",
|
|
138
|
+
"ig_vw": "1920",
|
|
139
|
+
"csrftoken": "",
|
|
140
|
+
"s_network": "",
|
|
141
|
+
"ds_user_id": "",
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
DEFAULT_USER_AGENT = (
|
|
145
|
+
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
|
|
146
|
+
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
|
147
|
+
"Chrome/124.0.0.0 Safari/537.36"
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
MEDIA_TYPES = {
|
|
151
|
+
1: InstagramMediaType.IMAGE,
|
|
152
|
+
2: InstagramMediaType.VIDEO,
|
|
153
|
+
8: InstagramMediaType.SIDECAR,
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
def __init__(
|
|
157
|
+
self,
|
|
158
|
+
*,
|
|
159
|
+
proxy: str | None = None,
|
|
160
|
+
cookie: dict[str, str] | None = None,
|
|
161
|
+
timeout: float = 30,
|
|
162
|
+
user_agent: str | None = None,
|
|
163
|
+
):
|
|
164
|
+
self.proxy = proxy
|
|
165
|
+
self.cookie = cookie or {}
|
|
166
|
+
self.timeout = timeout
|
|
167
|
+
self.user_agent = user_agent or self.DEFAULT_USER_AGENT
|
|
168
|
+
|
|
169
|
+
async def get_post(self, shortcode: str) -> InstagramPost:
|
|
170
|
+
media = await self.get_shortcode_media(shortcode)
|
|
171
|
+
return InstagramPost(media)
|
|
172
|
+
|
|
173
|
+
async def get_shortcode_media(self, shortcode: str) -> dict[str, Any]:
|
|
174
|
+
payload = await self._post_graphql(
|
|
175
|
+
doc_id=self.SHORTCODE_DOC_ID,
|
|
176
|
+
variables={
|
|
177
|
+
"shortcode": shortcode,
|
|
178
|
+
"__relay_internal__pv__PolarisAIGMMediaWebLabelEnabledrelayprovider": False,
|
|
179
|
+
},
|
|
180
|
+
)
|
|
181
|
+
|
|
182
|
+
media = self._extract_shortcode_media(payload)
|
|
183
|
+
if media is None:
|
|
184
|
+
raise InstagramAPIError("Fetching Post metadata failed.")
|
|
185
|
+
return media
|
|
186
|
+
|
|
187
|
+
async def _post_graphql(self, *, doc_id: str, variables: dict[str, Any]) -> dict[str, Any]:
|
|
188
|
+
async with self._new_client() as client:
|
|
189
|
+
await self._ensure_csrf_token(client)
|
|
190
|
+
data = {
|
|
191
|
+
"variables": json.dumps(variables, separators=(",", ":")),
|
|
192
|
+
"doc_id": doc_id,
|
|
193
|
+
"server_timestamps": "true",
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
try:
|
|
197
|
+
response = await client.post(self.GRAPHQL_URL, data=data, follow_redirects=False)
|
|
198
|
+
except httpx.HTTPError as exc:
|
|
199
|
+
raise InstagramAPIError(f"请求 Instagram GraphQL 失败: {exc}") from exc
|
|
200
|
+
|
|
201
|
+
if response.status_code != 200:
|
|
202
|
+
raise InstagramAPIError(f"Instagram GraphQL 返回 HTTP {response.status_code}: {response.text[:500]}")
|
|
203
|
+
|
|
204
|
+
try:
|
|
205
|
+
payload = response.json()
|
|
206
|
+
except ValueError as exc:
|
|
207
|
+
raise InstagramAPIError(f"Instagram GraphQL 返回非 JSON 响应: {response.text[:500]}") from exc
|
|
208
|
+
|
|
209
|
+
if payload.get("status") not in (None, "ok"):
|
|
210
|
+
raise InstagramAPIError(f"Instagram GraphQL 状态异常: {payload!r}")
|
|
211
|
+
return cast(dict[str, Any], payload)
|
|
212
|
+
|
|
213
|
+
def _extract_shortcode_media(self, payload: dict[str, Any]) -> dict[str, Any] | None:
|
|
214
|
+
data = payload.get("data")
|
|
215
|
+
if not isinstance(data, dict):
|
|
216
|
+
raise InstagramAPIError(f"响应缺少 data: {payload!r}")
|
|
217
|
+
|
|
218
|
+
old_media = data.get("xdt_shortcode_media")
|
|
219
|
+
if isinstance(old_media, dict):
|
|
220
|
+
return old_media
|
|
221
|
+
if old_media is None and "xdt_shortcode_media" in data:
|
|
222
|
+
return None
|
|
223
|
+
|
|
224
|
+
web_info = data.get("xdt_api__v1__media__shortcode__web_info")
|
|
225
|
+
if not isinstance(web_info, dict):
|
|
226
|
+
raise InstagramAPIError(f"响应缺少 shortcode web_info: {payload!r}")
|
|
227
|
+
|
|
228
|
+
items = web_info.get("items") or []
|
|
229
|
+
if not items:
|
|
230
|
+
return None
|
|
231
|
+
if not isinstance(items[0], dict):
|
|
232
|
+
raise InstagramAPIError(f"shortcode web_info.items[0] 类型异常: {type(items[0]).__name__}")
|
|
233
|
+
return self._convert_v1_media(items[0])
|
|
234
|
+
|
|
235
|
+
def _convert_v1_media(self, media: dict[str, Any]) -> dict[str, Any]:
|
|
236
|
+
media_type = self._media_type(media)
|
|
237
|
+
typename = self.MEDIA_TYPES.get(media_type, InstagramMediaType.IMAGE)
|
|
238
|
+
caption = media.get("caption")
|
|
239
|
+
caption_text = caption.get("text") if isinstance(caption, dict) else caption
|
|
240
|
+
image_candidate = self._first_image_candidate(media)
|
|
241
|
+
video_version = self._first_video_version(media)
|
|
242
|
+
|
|
243
|
+
node: dict[str, Any] = {
|
|
244
|
+
"shortcode": media.get("code", ""),
|
|
245
|
+
"id": media.get("pk", ""),
|
|
246
|
+
"__typename": typename,
|
|
247
|
+
"is_video": media_type == 2,
|
|
248
|
+
"taken_at_timestamp": media.get("taken_at"),
|
|
249
|
+
"edge_media_to_caption": {
|
|
250
|
+
"edges": [{"node": {"text": caption_text}}] if caption_text else [],
|
|
251
|
+
},
|
|
252
|
+
"edge_media_preview_like": {"count": media.get("like_count") or 0},
|
|
253
|
+
"edge_media_to_parent_comment": {
|
|
254
|
+
"count": media.get("comment_count") or 0,
|
|
255
|
+
"edges": [],
|
|
256
|
+
},
|
|
257
|
+
"owner": self._convert_owner(media.get("user")),
|
|
258
|
+
"dimensions": self._dimensions_from_candidate(image_candidate),
|
|
259
|
+
"display_url": image_candidate.get("url", ""),
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
for source_key, target_key in (
|
|
263
|
+
("title", "title"),
|
|
264
|
+
("has_liked", "viewer_has_liked"),
|
|
265
|
+
("accessibility_caption", "accessibility_caption"),
|
|
266
|
+
("location", "location"),
|
|
267
|
+
("video_duration", "video_duration"),
|
|
268
|
+
("view_count", "video_view_count"),
|
|
269
|
+
("play_count", "video_play_count"),
|
|
270
|
+
):
|
|
271
|
+
if media.get(source_key) is not None:
|
|
272
|
+
node[target_key] = media[source_key]
|
|
273
|
+
|
|
274
|
+
if video_version.get("url"):
|
|
275
|
+
node["video_url"] = video_version["url"]
|
|
276
|
+
|
|
277
|
+
carousel_media = media.get("carousel_media") or []
|
|
278
|
+
if carousel_media:
|
|
279
|
+
node["edge_sidecar_to_children"] = {
|
|
280
|
+
"edges": [{"node": self._convert_v1_sidecar_item(item)} for item in carousel_media],
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
return node
|
|
284
|
+
|
|
285
|
+
def _convert_v1_sidecar_item(self, item: dict[str, Any]) -> dict[str, Any]:
|
|
286
|
+
media_type = self._media_type(item)
|
|
287
|
+
typename = self.MEDIA_TYPES.get(media_type, InstagramMediaType.IMAGE)
|
|
288
|
+
image_candidate = self._first_image_candidate(item)
|
|
289
|
+
video_version = self._first_video_version(item)
|
|
290
|
+
is_video = media_type == 2
|
|
291
|
+
|
|
292
|
+
node: dict[str, Any] = {
|
|
293
|
+
"shortcode": item.get("code", ""),
|
|
294
|
+
"__typename": typename,
|
|
295
|
+
"is_video": is_video,
|
|
296
|
+
"display_url": image_candidate.get("url", ""),
|
|
297
|
+
"video_url": video_version.get("url") if is_video else None,
|
|
298
|
+
"dimensions": self._dimensions_from_candidate(image_candidate),
|
|
299
|
+
}
|
|
300
|
+
if item.get("accessibility_caption") is not None:
|
|
301
|
+
node["accessibility_caption"] = item["accessibility_caption"]
|
|
302
|
+
return node
|
|
303
|
+
|
|
304
|
+
@staticmethod
|
|
305
|
+
def _media_type(media: dict[str, Any]) -> int:
|
|
306
|
+
value = media.get("media_type")
|
|
307
|
+
return value if isinstance(value, int) else 0
|
|
308
|
+
|
|
309
|
+
@staticmethod
|
|
310
|
+
def _convert_owner(user: Any) -> dict[str, Any]:
|
|
311
|
+
if not isinstance(user, dict):
|
|
312
|
+
return {"id": "", "username": "", "full_name": ""}
|
|
313
|
+
return {
|
|
314
|
+
"id": user.get("pk", ""),
|
|
315
|
+
"username": user.get("username", ""),
|
|
316
|
+
"full_name": user.get("full_name", ""),
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
@staticmethod
|
|
320
|
+
def _first_image_candidate(media: dict[str, Any]) -> dict[str, Any]:
|
|
321
|
+
candidates = media.get("image_versions2", {}).get("candidates") or []
|
|
322
|
+
return candidates[0] if candidates and isinstance(candidates[0], dict) else {}
|
|
323
|
+
|
|
324
|
+
@staticmethod
|
|
325
|
+
def _first_video_version(media: dict[str, Any]) -> dict[str, Any]:
|
|
326
|
+
versions = media.get("video_versions") or []
|
|
327
|
+
return versions[0] if versions and isinstance(versions[0], dict) else {}
|
|
328
|
+
|
|
329
|
+
@staticmethod
|
|
330
|
+
def _dimensions_from_candidate(candidate: dict[str, Any]) -> dict[str, int]:
|
|
331
|
+
return {
|
|
332
|
+
"width": int(candidate.get("width") or 0),
|
|
333
|
+
"height": int(candidate.get("height") or 0),
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
async def _ensure_csrf_token(self, client: httpx.AsyncClient) -> None:
|
|
337
|
+
csrf_token = self._get_cookie_value(client, "csrftoken")
|
|
338
|
+
if not csrf_token:
|
|
339
|
+
try:
|
|
340
|
+
await client.get(self.INSTAGRAM_URL, follow_redirects=True)
|
|
341
|
+
except httpx.HTTPError as exc:
|
|
342
|
+
raise InstagramAPIError(f"获取 Instagram csrftoken 失败: {exc}") from exc
|
|
343
|
+
csrf_token = self._get_cookie_value(client, "csrftoken")
|
|
344
|
+
|
|
345
|
+
if not csrf_token:
|
|
346
|
+
raise InstagramAPIError("无法获取 Instagram csrftoken")
|
|
347
|
+
client.headers["x-csrftoken"] = csrf_token
|
|
348
|
+
|
|
349
|
+
@staticmethod
|
|
350
|
+
def _get_cookie_value(client: httpx.AsyncClient, name: str) -> str:
|
|
351
|
+
values = [cookie.value for cookie in client.cookies.jar if cookie.name == name and cookie.value]
|
|
352
|
+
return values[-1] if values else ""
|
|
353
|
+
|
|
354
|
+
def _new_client(self) -> httpx.AsyncClient:
|
|
355
|
+
cookies = self.DEFAULT_COOKIES | self.cookie
|
|
356
|
+
headers = {
|
|
357
|
+
"Accept": "*/*",
|
|
358
|
+
"Accept-Encoding": "gzip, deflate",
|
|
359
|
+
"Accept-Language": "en-US,en;q=0.8",
|
|
360
|
+
"Content-Type": "application/x-www-form-urlencoded",
|
|
361
|
+
"Referer": "https://www.instagram.com/",
|
|
362
|
+
"User-Agent": self.user_agent,
|
|
363
|
+
"authority": "www.instagram.com",
|
|
364
|
+
"scheme": "https",
|
|
365
|
+
}
|
|
366
|
+
return httpx.AsyncClient(
|
|
367
|
+
cookies=cookies,
|
|
368
|
+
headers=headers,
|
|
369
|
+
proxy=self.proxy,
|
|
370
|
+
timeout=self.timeout,
|
|
371
|
+
)
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
__all__ = [
|
|
375
|
+
"InstagramAPI",
|
|
376
|
+
"InstagramAPIError",
|
|
377
|
+
"InstagramMediaType",
|
|
378
|
+
"InstagramPost",
|
|
379
|
+
"InstagramSidecarNode",
|
|
380
|
+
]
|
|
@@ -100,7 +100,7 @@ class XHSAPI:
|
|
|
100
100
|
stream = selected_stream[0]
|
|
101
101
|
image = XHSMedia(
|
|
102
102
|
XHSMediaType.LIVE_PHOTO,
|
|
103
|
-
thumb_url=i["urlDefault"],
|
|
103
|
+
thumb_url=self.get_raw_image_url(i["urlDefault"]),
|
|
104
104
|
url=stream["masterUrl"],
|
|
105
105
|
width=i["width"],
|
|
106
106
|
height=i["height"],
|
|
@@ -108,7 +108,7 @@ class XHSAPI:
|
|
|
108
108
|
else:
|
|
109
109
|
image = XHSMedia(
|
|
110
110
|
XHSMediaType.IMAGE,
|
|
111
|
-
url=i["urlDefault"],
|
|
111
|
+
url=self.get_raw_image_url(i["urlDefault"]),
|
|
112
112
|
thumb_url=i["urlPre"],
|
|
113
113
|
width=i["width"],
|
|
114
114
|
height=i["height"],
|
|
@@ -120,6 +120,21 @@ class XHSAPI:
|
|
|
120
120
|
html = await self.__fetch_html(url)
|
|
121
121
|
return self.__parse(await self.__extract_data(html))
|
|
122
122
|
|
|
123
|
+
@staticmethod
|
|
124
|
+
def get_trace_id(img_url: str) -> str:
|
|
125
|
+
trace_id = img_url.split("/")[-1].split("!")[0]
|
|
126
|
+
if "spectrum" in img_url:
|
|
127
|
+
return "spectrum/" + trace_id
|
|
128
|
+
if "note_pre_post_uhdr" in img_url:
|
|
129
|
+
return "note_pre_post_uhdr/" + trace_id
|
|
130
|
+
if "notes_pre_post" in img_url:
|
|
131
|
+
return "notes_pre_post/" + trace_id
|
|
132
|
+
return trace_id
|
|
133
|
+
|
|
134
|
+
def get_raw_image_url(self, ime_url: str) -> str:
|
|
135
|
+
"""拼接无水印图片链接"""
|
|
136
|
+
return f"http://sns-img-hw.xhscdn.com/{self.get_trace_id(ime_url)}"
|
|
137
|
+
|
|
123
138
|
|
|
124
139
|
class XHSMediaType(Enum):
|
|
125
140
|
IMAGE = "image"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: parsehub
|
|
3
|
-
Version: 2.0.
|
|
3
|
+
Version: 2.0.34
|
|
4
4
|
Summary: 轻量、异步、开箱即用的社交媒体聚合解析库
|
|
5
5
|
Author-email: 梓澪 <zilingmio@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -24,7 +24,6 @@ Requires-Dist: tenacity>=8.5.0
|
|
|
24
24
|
Requires-Dist: urlextract>=1.9.0
|
|
25
25
|
Requires-Dist: yt-dlp[default]
|
|
26
26
|
Requires-Dist: lxml>=5.3.0
|
|
27
|
-
Requires-Dist: instaloader>=4.14
|
|
28
27
|
Requires-Dist: pydantic>=1.10.19
|
|
29
28
|
Requires-Dist: markdownify>=1.1.0
|
|
30
29
|
Requires-Dist: markdown>=3.7
|
|
@@ -1,79 +0,0 @@
|
|
|
1
|
-
import re
|
|
2
|
-
from collections.abc import Iterator
|
|
3
|
-
from typing import Any, NamedTuple
|
|
4
|
-
|
|
5
|
-
import requests
|
|
6
|
-
from instaloader import InstaloaderContext, InstaloaderException, Post
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
class MyPostSidecarNode(NamedTuple):
|
|
10
|
-
is_video: bool
|
|
11
|
-
display_url: str
|
|
12
|
-
video_url: str | None
|
|
13
|
-
width: int
|
|
14
|
-
height: int
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
class MyPost(Post):
|
|
18
|
-
def get_sidecar_nodes(self, start: int = 0, end: int = -1) -> Iterator[MyPostSidecarNode]: # type: ignore[override]
|
|
19
|
-
if self.typename == "GraphSidecar":
|
|
20
|
-
edges = self._field("edge_sidecar_to_children", "edges")
|
|
21
|
-
if end < 0:
|
|
22
|
-
end = len(edges) - 1
|
|
23
|
-
if start < 0:
|
|
24
|
-
start = len(edges) - 1
|
|
25
|
-
if any(edge["node"]["is_video"] and "video_url" not in edge["node"] for edge in edges[start : (end + 1)]):
|
|
26
|
-
# video_url is only present in full metadata, issue #558.
|
|
27
|
-
edges = self._full_metadata["edge_sidecar_to_children"]["edges"]
|
|
28
|
-
for idx, edge in enumerate(edges):
|
|
29
|
-
if start <= idx <= end:
|
|
30
|
-
node = edge["node"]
|
|
31
|
-
is_video = node["is_video"]
|
|
32
|
-
display_url = node["display_url"]
|
|
33
|
-
dimensions = node["dimensions"]
|
|
34
|
-
width = dimensions["width"]
|
|
35
|
-
height = dimensions["height"]
|
|
36
|
-
|
|
37
|
-
if not is_video and self._context.iphone_support and self._context.is_logged_in:
|
|
38
|
-
try:
|
|
39
|
-
carousel_media = self._iphone_struct["carousel_media"]
|
|
40
|
-
orig_url = carousel_media[idx]["image_versions2"]["candidates"][0]["url"]
|
|
41
|
-
display_url = re.sub(r"([?&])se=\d+&?", r"\1", orig_url).rstrip("&")
|
|
42
|
-
except (InstaloaderException, KeyError, IndexError) as err:
|
|
43
|
-
self._context.error(f"Unable to fetch high quality image version of {self}: {err}")
|
|
44
|
-
yield MyPostSidecarNode(
|
|
45
|
-
is_video=is_video,
|
|
46
|
-
display_url=display_url,
|
|
47
|
-
video_url=node["video_url"] if is_video else None,
|
|
48
|
-
width=width,
|
|
49
|
-
height=height,
|
|
50
|
-
)
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
class MyInstaloaderContext(InstaloaderContext):
|
|
54
|
-
"""
|
|
55
|
-
支持自定义代理
|
|
56
|
-
"""
|
|
57
|
-
|
|
58
|
-
def __init__(self, proxy: str | None = None, cookie: dict | None = None):
|
|
59
|
-
self.proxy: dict[str, str | None] = {"http": proxy, "https": proxy}
|
|
60
|
-
self.cookie = cookie
|
|
61
|
-
super().__init__()
|
|
62
|
-
|
|
63
|
-
def get_anonymous_session(self) -> requests.Session:
|
|
64
|
-
session = super().get_anonymous_session()
|
|
65
|
-
if self.proxy:
|
|
66
|
-
session.proxies = {k: v for k, v in self.proxy.items() if v is not None}
|
|
67
|
-
session.trust_env = False
|
|
68
|
-
return session
|
|
69
|
-
|
|
70
|
-
def get_json(self, *args: Any, **kwargs: Any) -> Any:
|
|
71
|
-
session = kwargs.get("session")
|
|
72
|
-
if isinstance(session, requests.Session):
|
|
73
|
-
if self.proxy:
|
|
74
|
-
session.proxies = {k: v for k, v in self.proxy.items() if v is not None}
|
|
75
|
-
session.trust_env = False
|
|
76
|
-
if self.cookie:
|
|
77
|
-
session.cookies.update(self.cookie)
|
|
78
|
-
|
|
79
|
-
return super().get_json(*args, **kwargs)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|