parsehub 2.0.33__tar.gz → 2.0.35__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {parsehub-2.0.33/src/parsehub.egg-info → parsehub-2.0.35}/PKG-INFO +1 -2
- {parsehub-2.0.33 → parsehub-2.0.35}/pyproject.toml +1 -2
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/instagram.py +12 -30
- parsehub-2.0.35/src/parsehub/provider_api/instagram.py +380 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/provider_api/xhs.py +8 -8
- {parsehub-2.0.33 → parsehub-2.0.35/src/parsehub.egg-info}/PKG-INFO +1 -2
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub.egg-info/SOURCES.txt +2 -1
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub.egg-info/requires.txt +0 -1
- {parsehub-2.0.33 → parsehub-2.0.35}/test/test_core_offline.py +1 -0
- parsehub-2.0.35/test/test_downloader.py +161 -0
- parsehub-2.0.33/src/parsehub/provider_api/instagram.py +0 -79
- {parsehub-2.0.33 → parsehub-2.0.35}/LICENSE +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/README.md +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/setup.cfg +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/__init__.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/__init__.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/cli.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/cli_config.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/config/__init__.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/config/config.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/errors.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/__init__.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/base/__init__.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/base/base.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/base/ytdlp.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/__init__.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/bilibili.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/coolapk.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/douyin.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/facebook.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/kuaishou.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/pipix.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/snapchat.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/threads.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/tieba.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/tiktok.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/twitter.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/weibo.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/weixin.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/xhs.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/xiaoheihe.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/youtube.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/parsers/parser/zuiyou.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/provider_api/__init__.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/provider_api/bilibili.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/provider_api/coolapk.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/provider_api/douyin.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/provider_api/kuaishou.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/provider_api/pipix.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/provider_api/threads.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/provider_api/tieba.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/provider_api/tiktok.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/provider_api/twitter.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/provider_api/weibo.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/provider_api/weixin.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/provider_api/xiaoheihe.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/provider_api/zuiyou.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/types/__init__.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/types/callback.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/types/media_file.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/types/media_ref.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/types/platform.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/types/post.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/types/result.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/utils/downloader.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/utils/helpers.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub/utils/media_info.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub.egg-info/dependency_links.txt +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub.egg-info/entry_points.txt +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/src/parsehub.egg-info/top_level.txt +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/test/test_cli.py +0 -0
- {parsehub-2.0.33 → parsehub-2.0.35}/test/test_cli_config.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: parsehub
|
|
3
|
-
Version: 2.0.
|
|
3
|
+
Version: 2.0.35
|
|
4
4
|
Summary: 轻量、异步、开箱即用的社交媒体聚合解析库
|
|
5
5
|
Author-email: 梓澪 <zilingmio@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -24,7 +24,6 @@ Requires-Dist: tenacity>=8.5.0
|
|
|
24
24
|
Requires-Dist: urlextract>=1.9.0
|
|
25
25
|
Requires-Dist: yt-dlp[default]
|
|
26
26
|
Requires-Dist: lxml>=5.3.0
|
|
27
|
-
Requires-Dist: instaloader>=4.14
|
|
28
27
|
Requires-Dist: pydantic>=1.10.19
|
|
29
28
|
Requires-Dist: markdownify>=1.1.0
|
|
30
29
|
Requires-Dist: markdown>=3.7
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "parsehub"
|
|
3
|
-
version = "2.0.
|
|
3
|
+
version = "2.0.35"
|
|
4
4
|
description = "轻量、异步、开箱即用的社交媒体聚合解析库"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.12.0"
|
|
@@ -27,7 +27,6 @@ dependencies = [
|
|
|
27
27
|
"urlextract>=1.9.0",
|
|
28
28
|
"yt-dlp[default]",
|
|
29
29
|
"lxml>=5.3.0",
|
|
30
|
-
"instaloader>=4.14",
|
|
31
30
|
"pydantic>=1.10.19",
|
|
32
31
|
"markdownify>=1.1.0",
|
|
33
32
|
"markdown>=3.7",
|
|
@@ -1,10 +1,6 @@
|
|
|
1
|
-
import asyncio
|
|
2
1
|
import re
|
|
3
|
-
from typing import cast
|
|
4
2
|
|
|
5
|
-
from
|
|
6
|
-
|
|
7
|
-
from ...provider_api.instagram import MyInstaloaderContext, MyPost
|
|
3
|
+
from ...provider_api.instagram import InstagramAPI, InstagramAPIError, InstagramMediaType, InstagramPost
|
|
8
4
|
from ...types import ImageParseResult, ImageRef, MultimediaParseResult, ParseError, Platform, VideoParseResult, VideoRef
|
|
9
5
|
from ...utils.helpers import SecretCookie
|
|
10
6
|
from ..base.base import BaseParser
|
|
@@ -13,7 +9,7 @@ from ..base.base import BaseParser
|
|
|
13
9
|
class InstagramParser(BaseParser):
|
|
14
10
|
__platform__ = Platform.INSTAGRAM
|
|
15
11
|
__supported_type__ = ["视频", "图文"]
|
|
16
|
-
__match__ = r"^(http(s)?://)(www\.|)instagram\.com/(p|
|
|
12
|
+
__match__ = r"^(http(s)?://)(www\.|)instagram\.com/(p|reels|share|.*/p)/.*"
|
|
17
13
|
__redirect_keywords__ = ["share"]
|
|
18
14
|
|
|
19
15
|
async def _do_parse(self, raw_url: str) -> VideoParseResult | ImageParseResult | MultimediaParseResult:
|
|
@@ -23,14 +19,10 @@ class InstagramParser(BaseParser):
|
|
|
23
19
|
|
|
24
20
|
post = await self._parse(raw_url, shortcode)
|
|
25
21
|
|
|
26
|
-
|
|
27
|
-
dimensions: dict = post._field("dimensions")
|
|
28
|
-
except KeyError:
|
|
29
|
-
dimensions = {}
|
|
30
|
-
width, height = dimensions.get("width", 0) or 0, dimensions.get("height", 0) or 0
|
|
22
|
+
width, height = post.width, post.height
|
|
31
23
|
|
|
32
24
|
match post.typename:
|
|
33
|
-
case
|
|
25
|
+
case InstagramMediaType.SIDECAR:
|
|
34
26
|
media = [
|
|
35
27
|
VideoRef(url=i.video_url, thumb_url=i.display_url, width=i.width, height=i.height)
|
|
36
28
|
if i.is_video and i.video_url
|
|
@@ -38,11 +30,11 @@ class InstagramParser(BaseParser):
|
|
|
38
30
|
for i in post.get_sidecar_nodes()
|
|
39
31
|
]
|
|
40
32
|
return MultimediaParseResult(media=media, title=post.title, content=post.caption)
|
|
41
|
-
case
|
|
33
|
+
case InstagramMediaType.IMAGE:
|
|
42
34
|
return ImageParseResult(
|
|
43
35
|
photo=[ImageRef(url=post.url, width=width, height=height)], title=post.title, content=post.caption
|
|
44
36
|
)
|
|
45
|
-
case
|
|
37
|
+
case InstagramMediaType.VIDEO:
|
|
46
38
|
return VideoParseResult(
|
|
47
39
|
video=VideoRef(
|
|
48
40
|
url=post.video_url or post.url,
|
|
@@ -57,19 +49,11 @@ class InstagramParser(BaseParser):
|
|
|
57
49
|
case _:
|
|
58
50
|
raise ParseError("不支持的类型")
|
|
59
51
|
|
|
60
|
-
async def _parse(self, url: str, shortcode: str, cookie: SecretCookie | None = None) ->
|
|
52
|
+
async def _parse(self, url: str, shortcode: str, cookie: SecretCookie | None = None) -> InstagramPost:
|
|
61
53
|
try:
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
MyInstaloaderContext(self.proxy, cookie.get_value() if cookie else None),
|
|
66
|
-
shortcode,
|
|
67
|
-
),
|
|
68
|
-
30,
|
|
69
|
-
)
|
|
70
|
-
except TimeoutError as e:
|
|
71
|
-
raise ParseError("解析超时") from e
|
|
72
|
-
except BadResponseException as e:
|
|
54
|
+
api = InstagramAPI(proxy=self.proxy, cookie=cookie.get_value() if cookie else None, timeout=30)
|
|
55
|
+
return await api.get_post(shortcode)
|
|
56
|
+
except InstagramAPIError as e:
|
|
73
57
|
match str(e):
|
|
74
58
|
case "Fetching Post metadata failed.":
|
|
75
59
|
if self.cookie and cookie is None:
|
|
@@ -83,14 +67,12 @@ class InstagramParser(BaseParser):
|
|
|
83
67
|
text = f"Instagram 账号可能已被封禁\n\n使用的Cookie: {cookie}"
|
|
84
68
|
else:
|
|
85
69
|
text = str(e)
|
|
86
|
-
raise ParseError(f"
|
|
87
|
-
else:
|
|
88
|
-
return cast(MyPost, post)
|
|
70
|
+
raise ParseError(f"无法获取帖子内容(可能为私人内容): {text}") from e
|
|
89
71
|
|
|
90
72
|
@staticmethod
|
|
91
73
|
def get_short_code(url: str) -> str | None:
|
|
92
74
|
url = url.removesuffix("/")
|
|
93
|
-
shortcode = re.search(r"/(share|p|
|
|
75
|
+
shortcode = re.search(r"/(share|p|reels|.*/p)/(.*)", url)
|
|
94
76
|
return shortcode.group(2).split("/")[0] if shortcode else None
|
|
95
77
|
|
|
96
78
|
|
|
@@ -0,0 +1,380 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from collections.abc import Iterator
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from enum import StrEnum
|
|
7
|
+
from typing import Any, cast
|
|
8
|
+
|
|
9
|
+
import httpx
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class InstagramAPIError(RuntimeError):
|
|
13
|
+
"""Instagram 接口请求或响应解析失败。"""
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class InstagramMediaType(StrEnum):
|
|
17
|
+
IMAGE = "GraphImage"
|
|
18
|
+
VIDEO = "GraphVideo"
|
|
19
|
+
SIDECAR = "GraphSidecar"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(slots=True)
|
|
23
|
+
class InstagramSidecarNode:
|
|
24
|
+
is_video: bool
|
|
25
|
+
display_url: str
|
|
26
|
+
video_url: str | None
|
|
27
|
+
width: int
|
|
28
|
+
height: int
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class InstagramPost:
|
|
32
|
+
"""轻量版 Instagram Post,只保留当前项目解析帖子需要的字段。"""
|
|
33
|
+
|
|
34
|
+
_XDT_TYPES = {
|
|
35
|
+
"XDTGraphImage": InstagramMediaType.IMAGE,
|
|
36
|
+
"XDTGraphVideo": InstagramMediaType.VIDEO,
|
|
37
|
+
"XDTGraphSidecar": InstagramMediaType.SIDECAR,
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
def __init__(self, node: dict[str, Any]):
|
|
41
|
+
self._node = node
|
|
42
|
+
self._normalize_typename()
|
|
43
|
+
|
|
44
|
+
def _normalize_typename(self) -> None:
|
|
45
|
+
typename = self._node.get("__typename")
|
|
46
|
+
if typename in self._XDT_TYPES:
|
|
47
|
+
self._node["__typename"] = self._XDT_TYPES[typename]
|
|
48
|
+
|
|
49
|
+
def _field(self, *keys: str) -> Any:
|
|
50
|
+
value: Any = self._node
|
|
51
|
+
for key in keys:
|
|
52
|
+
value = value[key]
|
|
53
|
+
return value
|
|
54
|
+
|
|
55
|
+
@property
|
|
56
|
+
def shortcode(self) -> str:
|
|
57
|
+
return str(self._node.get("shortcode") or self._node["code"])
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def typename(self) -> InstagramMediaType:
|
|
61
|
+
return InstagramMediaType(self._field("__typename"))
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def is_video(self) -> bool:
|
|
65
|
+
return bool(self._field("is_video"))
|
|
66
|
+
|
|
67
|
+
@property
|
|
68
|
+
def title(self) -> str | None:
|
|
69
|
+
return self._node.get("title")
|
|
70
|
+
|
|
71
|
+
@property
|
|
72
|
+
def caption(self) -> str | None:
|
|
73
|
+
caption_edges = self._node.get("edge_media_to_caption", {}).get("edges") or []
|
|
74
|
+
if caption_edges:
|
|
75
|
+
if text := caption_edges[0].get("node", {}).get("text"):
|
|
76
|
+
return str(text)
|
|
77
|
+
return None
|
|
78
|
+
return self._node.get("caption")
|
|
79
|
+
|
|
80
|
+
@property
|
|
81
|
+
def url(self) -> str:
|
|
82
|
+
return str(self._node.get("display_url") or self._node["display_src"])
|
|
83
|
+
|
|
84
|
+
@property
|
|
85
|
+
def video_url(self) -> str | None:
|
|
86
|
+
if not self.is_video:
|
|
87
|
+
return None
|
|
88
|
+
return self._node.get("video_url")
|
|
89
|
+
|
|
90
|
+
@property
|
|
91
|
+
def video_duration(self) -> float | None:
|
|
92
|
+
value = self._node.get("video_duration")
|
|
93
|
+
return float(value) if value is not None else None
|
|
94
|
+
|
|
95
|
+
@property
|
|
96
|
+
def width(self) -> int:
|
|
97
|
+
return int(self._node.get("dimensions", {}).get("width") or 0)
|
|
98
|
+
|
|
99
|
+
@property
|
|
100
|
+
def height(self) -> int:
|
|
101
|
+
return int(self._node.get("dimensions", {}).get("height") or 0)
|
|
102
|
+
|
|
103
|
+
def get_sidecar_nodes(self, start: int = 0, end: int = -1) -> Iterator[InstagramSidecarNode]:
|
|
104
|
+
if self.typename is not InstagramMediaType.SIDECAR:
|
|
105
|
+
return
|
|
106
|
+
|
|
107
|
+
edges = self._field("edge_sidecar_to_children", "edges")
|
|
108
|
+
if end < 0:
|
|
109
|
+
end = len(edges) - 1
|
|
110
|
+
if start < 0:
|
|
111
|
+
start = len(edges) - 1
|
|
112
|
+
|
|
113
|
+
for idx, edge in enumerate(edges):
|
|
114
|
+
if not start <= idx <= end:
|
|
115
|
+
continue
|
|
116
|
+
|
|
117
|
+
node = edge["node"]
|
|
118
|
+
dimensions = node.get("dimensions", {})
|
|
119
|
+
is_video = bool(node.get("is_video"))
|
|
120
|
+
yield InstagramSidecarNode(
|
|
121
|
+
is_video=is_video,
|
|
122
|
+
display_url=node.get("display_url") or node.get("display_src") or "",
|
|
123
|
+
video_url=node.get("video_url") if is_video else None,
|
|
124
|
+
width=int(dimensions.get("width") or 0),
|
|
125
|
+
height=int(dimensions.get("height") or 0),
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
class InstagramAPI:
|
|
130
|
+
GRAPHQL_URL = "https://www.instagram.com/graphql/query"
|
|
131
|
+
INSTAGRAM_URL = "https://www.instagram.com/"
|
|
132
|
+
SHORTCODE_DOC_ID = "27128499623469141"
|
|
133
|
+
|
|
134
|
+
DEFAULT_COOKIES = {
|
|
135
|
+
"sessionid": "",
|
|
136
|
+
"mid": "",
|
|
137
|
+
"ig_pr": "1",
|
|
138
|
+
"ig_vw": "1920",
|
|
139
|
+
"csrftoken": "",
|
|
140
|
+
"s_network": "",
|
|
141
|
+
"ds_user_id": "",
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
DEFAULT_USER_AGENT = (
|
|
145
|
+
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
|
|
146
|
+
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
|
147
|
+
"Chrome/124.0.0.0 Safari/537.36"
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
MEDIA_TYPES = {
|
|
151
|
+
1: InstagramMediaType.IMAGE,
|
|
152
|
+
2: InstagramMediaType.VIDEO,
|
|
153
|
+
8: InstagramMediaType.SIDECAR,
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
def __init__(
|
|
157
|
+
self,
|
|
158
|
+
*,
|
|
159
|
+
proxy: str | None = None,
|
|
160
|
+
cookie: dict[str, str] | None = None,
|
|
161
|
+
timeout: float = 30,
|
|
162
|
+
user_agent: str | None = None,
|
|
163
|
+
):
|
|
164
|
+
self.proxy = proxy
|
|
165
|
+
self.cookie = cookie or {}
|
|
166
|
+
self.timeout = timeout
|
|
167
|
+
self.user_agent = user_agent or self.DEFAULT_USER_AGENT
|
|
168
|
+
|
|
169
|
+
async def get_post(self, shortcode: str) -> InstagramPost:
|
|
170
|
+
media = await self.get_shortcode_media(shortcode)
|
|
171
|
+
return InstagramPost(media)
|
|
172
|
+
|
|
173
|
+
async def get_shortcode_media(self, shortcode: str) -> dict[str, Any]:
|
|
174
|
+
payload = await self._post_graphql(
|
|
175
|
+
doc_id=self.SHORTCODE_DOC_ID,
|
|
176
|
+
variables={
|
|
177
|
+
"shortcode": shortcode,
|
|
178
|
+
"__relay_internal__pv__PolarisAIGMMediaWebLabelEnabledrelayprovider": False,
|
|
179
|
+
},
|
|
180
|
+
)
|
|
181
|
+
|
|
182
|
+
media = self._extract_shortcode_media(payload)
|
|
183
|
+
if media is None:
|
|
184
|
+
raise InstagramAPIError("Fetching Post metadata failed.")
|
|
185
|
+
return media
|
|
186
|
+
|
|
187
|
+
async def _post_graphql(self, *, doc_id: str, variables: dict[str, Any]) -> dict[str, Any]:
|
|
188
|
+
async with self._new_client() as client:
|
|
189
|
+
await self._ensure_csrf_token(client)
|
|
190
|
+
data = {
|
|
191
|
+
"variables": json.dumps(variables, separators=(",", ":")),
|
|
192
|
+
"doc_id": doc_id,
|
|
193
|
+
"server_timestamps": "true",
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
try:
|
|
197
|
+
response = await client.post(self.GRAPHQL_URL, data=data, follow_redirects=False)
|
|
198
|
+
except httpx.HTTPError as exc:
|
|
199
|
+
raise InstagramAPIError(f"请求 Instagram GraphQL 失败: {exc}") from exc
|
|
200
|
+
|
|
201
|
+
if response.status_code != 200:
|
|
202
|
+
raise InstagramAPIError(f"Instagram GraphQL 返回 HTTP {response.status_code}: {response.text[:500]}")
|
|
203
|
+
|
|
204
|
+
try:
|
|
205
|
+
payload = response.json()
|
|
206
|
+
except ValueError as exc:
|
|
207
|
+
raise InstagramAPIError(f"Instagram GraphQL 返回非 JSON 响应: {response.text[:500]}") from exc
|
|
208
|
+
|
|
209
|
+
if payload.get("status") not in (None, "ok"):
|
|
210
|
+
raise InstagramAPIError(f"Instagram GraphQL 状态异常: {payload!r}")
|
|
211
|
+
return cast(dict[str, Any], payload)
|
|
212
|
+
|
|
213
|
+
def _extract_shortcode_media(self, payload: dict[str, Any]) -> dict[str, Any] | None:
|
|
214
|
+
data = payload.get("data")
|
|
215
|
+
if not isinstance(data, dict):
|
|
216
|
+
raise InstagramAPIError(f"响应缺少 data: {payload!r}")
|
|
217
|
+
|
|
218
|
+
old_media = data.get("xdt_shortcode_media")
|
|
219
|
+
if isinstance(old_media, dict):
|
|
220
|
+
return old_media
|
|
221
|
+
if old_media is None and "xdt_shortcode_media" in data:
|
|
222
|
+
return None
|
|
223
|
+
|
|
224
|
+
web_info = data.get("xdt_api__v1__media__shortcode__web_info")
|
|
225
|
+
if not isinstance(web_info, dict):
|
|
226
|
+
raise InstagramAPIError(f"响应缺少 shortcode web_info: {payload!r}")
|
|
227
|
+
|
|
228
|
+
items = web_info.get("items") or []
|
|
229
|
+
if not items:
|
|
230
|
+
return None
|
|
231
|
+
if not isinstance(items[0], dict):
|
|
232
|
+
raise InstagramAPIError(f"shortcode web_info.items[0] 类型异常: {type(items[0]).__name__}")
|
|
233
|
+
return self._convert_v1_media(items[0])
|
|
234
|
+
|
|
235
|
+
def _convert_v1_media(self, media: dict[str, Any]) -> dict[str, Any]:
|
|
236
|
+
media_type = self._media_type(media)
|
|
237
|
+
typename = self.MEDIA_TYPES.get(media_type, InstagramMediaType.IMAGE)
|
|
238
|
+
caption = media.get("caption")
|
|
239
|
+
caption_text = caption.get("text") if isinstance(caption, dict) else caption
|
|
240
|
+
image_candidate = self._first_image_candidate(media)
|
|
241
|
+
video_version = self._first_video_version(media)
|
|
242
|
+
|
|
243
|
+
node: dict[str, Any] = {
|
|
244
|
+
"shortcode": media.get("code", ""),
|
|
245
|
+
"id": media.get("pk", ""),
|
|
246
|
+
"__typename": typename,
|
|
247
|
+
"is_video": media_type == 2,
|
|
248
|
+
"taken_at_timestamp": media.get("taken_at"),
|
|
249
|
+
"edge_media_to_caption": {
|
|
250
|
+
"edges": [{"node": {"text": caption_text}}] if caption_text else [],
|
|
251
|
+
},
|
|
252
|
+
"edge_media_preview_like": {"count": media.get("like_count") or 0},
|
|
253
|
+
"edge_media_to_parent_comment": {
|
|
254
|
+
"count": media.get("comment_count") or 0,
|
|
255
|
+
"edges": [],
|
|
256
|
+
},
|
|
257
|
+
"owner": self._convert_owner(media.get("user")),
|
|
258
|
+
"dimensions": self._dimensions_from_candidate(image_candidate),
|
|
259
|
+
"display_url": image_candidate.get("url", ""),
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
for source_key, target_key in (
|
|
263
|
+
("title", "title"),
|
|
264
|
+
("has_liked", "viewer_has_liked"),
|
|
265
|
+
("accessibility_caption", "accessibility_caption"),
|
|
266
|
+
("location", "location"),
|
|
267
|
+
("video_duration", "video_duration"),
|
|
268
|
+
("view_count", "video_view_count"),
|
|
269
|
+
("play_count", "video_play_count"),
|
|
270
|
+
):
|
|
271
|
+
if media.get(source_key) is not None:
|
|
272
|
+
node[target_key] = media[source_key]
|
|
273
|
+
|
|
274
|
+
if video_version.get("url"):
|
|
275
|
+
node["video_url"] = video_version["url"]
|
|
276
|
+
|
|
277
|
+
carousel_media = media.get("carousel_media") or []
|
|
278
|
+
if carousel_media:
|
|
279
|
+
node["edge_sidecar_to_children"] = {
|
|
280
|
+
"edges": [{"node": self._convert_v1_sidecar_item(item)} for item in carousel_media],
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
return node
|
|
284
|
+
|
|
285
|
+
def _convert_v1_sidecar_item(self, item: dict[str, Any]) -> dict[str, Any]:
|
|
286
|
+
media_type = self._media_type(item)
|
|
287
|
+
typename = self.MEDIA_TYPES.get(media_type, InstagramMediaType.IMAGE)
|
|
288
|
+
image_candidate = self._first_image_candidate(item)
|
|
289
|
+
video_version = self._first_video_version(item)
|
|
290
|
+
is_video = media_type == 2
|
|
291
|
+
|
|
292
|
+
node: dict[str, Any] = {
|
|
293
|
+
"shortcode": item.get("code", ""),
|
|
294
|
+
"__typename": typename,
|
|
295
|
+
"is_video": is_video,
|
|
296
|
+
"display_url": image_candidate.get("url", ""),
|
|
297
|
+
"video_url": video_version.get("url") if is_video else None,
|
|
298
|
+
"dimensions": self._dimensions_from_candidate(image_candidate),
|
|
299
|
+
}
|
|
300
|
+
if item.get("accessibility_caption") is not None:
|
|
301
|
+
node["accessibility_caption"] = item["accessibility_caption"]
|
|
302
|
+
return node
|
|
303
|
+
|
|
304
|
+
@staticmethod
|
|
305
|
+
def _media_type(media: dict[str, Any]) -> int:
|
|
306
|
+
value = media.get("media_type")
|
|
307
|
+
return value if isinstance(value, int) else 0
|
|
308
|
+
|
|
309
|
+
@staticmethod
|
|
310
|
+
def _convert_owner(user: Any) -> dict[str, Any]:
|
|
311
|
+
if not isinstance(user, dict):
|
|
312
|
+
return {"id": "", "username": "", "full_name": ""}
|
|
313
|
+
return {
|
|
314
|
+
"id": user.get("pk", ""),
|
|
315
|
+
"username": user.get("username", ""),
|
|
316
|
+
"full_name": user.get("full_name", ""),
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
@staticmethod
|
|
320
|
+
def _first_image_candidate(media: dict[str, Any]) -> dict[str, Any]:
|
|
321
|
+
candidates = media.get("image_versions2", {}).get("candidates") or []
|
|
322
|
+
return candidates[0] if candidates and isinstance(candidates[0], dict) else {}
|
|
323
|
+
|
|
324
|
+
@staticmethod
|
|
325
|
+
def _first_video_version(media: dict[str, Any]) -> dict[str, Any]:
|
|
326
|
+
versions = media.get("video_versions") or []
|
|
327
|
+
return versions[0] if versions and isinstance(versions[0], dict) else {}
|
|
328
|
+
|
|
329
|
+
@staticmethod
|
|
330
|
+
def _dimensions_from_candidate(candidate: dict[str, Any]) -> dict[str, int]:
|
|
331
|
+
return {
|
|
332
|
+
"width": int(candidate.get("width") or 0),
|
|
333
|
+
"height": int(candidate.get("height") or 0),
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
async def _ensure_csrf_token(self, client: httpx.AsyncClient) -> None:
|
|
337
|
+
csrf_token = self._get_cookie_value(client, "csrftoken")
|
|
338
|
+
if not csrf_token:
|
|
339
|
+
try:
|
|
340
|
+
await client.get(self.INSTAGRAM_URL, follow_redirects=True)
|
|
341
|
+
except httpx.HTTPError as exc:
|
|
342
|
+
raise InstagramAPIError(f"获取 Instagram csrftoken 失败: {exc}") from exc
|
|
343
|
+
csrf_token = self._get_cookie_value(client, "csrftoken")
|
|
344
|
+
|
|
345
|
+
if not csrf_token:
|
|
346
|
+
raise InstagramAPIError("无法获取 Instagram csrftoken")
|
|
347
|
+
client.headers["x-csrftoken"] = csrf_token
|
|
348
|
+
|
|
349
|
+
@staticmethod
|
|
350
|
+
def _get_cookie_value(client: httpx.AsyncClient, name: str) -> str:
|
|
351
|
+
values = [cookie.value for cookie in client.cookies.jar if cookie.name == name and cookie.value]
|
|
352
|
+
return values[-1] if values else ""
|
|
353
|
+
|
|
354
|
+
def _new_client(self) -> httpx.AsyncClient:
|
|
355
|
+
cookies = self.DEFAULT_COOKIES | self.cookie
|
|
356
|
+
headers = {
|
|
357
|
+
"Accept": "*/*",
|
|
358
|
+
"Accept-Encoding": "gzip, deflate",
|
|
359
|
+
"Accept-Language": "en-US,en;q=0.8",
|
|
360
|
+
"Content-Type": "application/x-www-form-urlencoded",
|
|
361
|
+
"Referer": "https://www.instagram.com/",
|
|
362
|
+
"User-Agent": self.user_agent,
|
|
363
|
+
"authority": "www.instagram.com",
|
|
364
|
+
"scheme": "https",
|
|
365
|
+
}
|
|
366
|
+
return httpx.AsyncClient(
|
|
367
|
+
cookies=cookies,
|
|
368
|
+
headers=headers,
|
|
369
|
+
proxy=self.proxy,
|
|
370
|
+
timeout=self.timeout,
|
|
371
|
+
)
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
__all__ = [
|
|
375
|
+
"InstagramAPI",
|
|
376
|
+
"InstagramAPIError",
|
|
377
|
+
"InstagramMediaType",
|
|
378
|
+
"InstagramPost",
|
|
379
|
+
"InstagramSidecarNode",
|
|
380
|
+
]
|
|
@@ -122,14 +122,14 @@ class XHSAPI:
|
|
|
122
122
|
|
|
123
123
|
@staticmethod
|
|
124
124
|
def get_trace_id(img_url: str) -> str:
|
|
125
|
-
|
|
126
|
-
if
|
|
127
|
-
return "
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
if
|
|
131
|
-
return
|
|
132
|
-
return
|
|
125
|
+
match = re.search(r"/(spectrum|note_pre_post_uhdr|notes_pre_post|notes_uhdr)/([^/!]+)(?:!.*)?$", img_url)
|
|
126
|
+
if match:
|
|
127
|
+
return f"{match.group(1)}/{match.group(2)}"
|
|
128
|
+
|
|
129
|
+
match = re.search(r"/([^/!]+)(?:!.*)?$", img_url)
|
|
130
|
+
if match:
|
|
131
|
+
return match.group(1)
|
|
132
|
+
return img_url
|
|
133
133
|
|
|
134
134
|
def get_raw_image_url(self, ime_url: str) -> str:
|
|
135
135
|
"""拼接无水印图片链接"""
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: parsehub
|
|
3
|
-
Version: 2.0.
|
|
3
|
+
Version: 2.0.35
|
|
4
4
|
Summary: 轻量、异步、开箱即用的社交媒体聚合解析库
|
|
5
5
|
Author-email: 梓澪 <zilingmio@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -24,7 +24,6 @@ Requires-Dist: tenacity>=8.5.0
|
|
|
24
24
|
Requires-Dist: urlextract>=1.9.0
|
|
25
25
|
Requires-Dist: yt-dlp[default]
|
|
26
26
|
Requires-Dist: lxml>=5.3.0
|
|
27
|
-
Requires-Dist: instaloader>=4.14
|
|
28
27
|
Requires-Dist: pydantic>=1.10.19
|
|
29
28
|
Requires-Dist: markdownify>=1.1.0
|
|
30
29
|
Requires-Dist: markdown>=3.7
|
|
@@ -238,6 +238,7 @@ class TestPlatformUrlMatching(unittest.TestCase):
|
|
|
238
238
|
"https://www.instagram.com/share/BAexample/",
|
|
239
239
|
"https://www.instagram.com/user.name/p/C0example/",
|
|
240
240
|
"https://www.instagram.com/user.name/reel/C0example/",
|
|
241
|
+
"https://www.instagram.com/reels/DaGI8bPS3ed/",
|
|
241
242
|
],
|
|
242
243
|
Platform.KUAISHOU: [
|
|
243
244
|
"https://www.kuaishou.com/short-video/3xexample",
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
import contextlib
|
|
2
|
+
import threading
|
|
3
|
+
import unittest
|
|
4
|
+
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from tempfile import TemporaryDirectory
|
|
7
|
+
from typing import ClassVar
|
|
8
|
+
|
|
9
|
+
from parsehub.errors import DownloadError
|
|
10
|
+
from parsehub.utils.downloader import download
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class RangeTestHandler(BaseHTTPRequestHandler):
|
|
14
|
+
content: ClassVar[bytes] = b""
|
|
15
|
+
support_range: ClassVar[bool] = True
|
|
16
|
+
fail_all: ClassVar[bool] = False
|
|
17
|
+
requests: ClassVar[list[tuple[str, str | None]]] = []
|
|
18
|
+
|
|
19
|
+
def log_message(self, format: str, *args: object) -> None:
|
|
20
|
+
return
|
|
21
|
+
|
|
22
|
+
def do_HEAD(self) -> None:
|
|
23
|
+
self.__class__.requests.append(("HEAD", self.headers.get("Range")))
|
|
24
|
+
if self.fail_all:
|
|
25
|
+
self.send_response(500)
|
|
26
|
+
self.end_headers()
|
|
27
|
+
return
|
|
28
|
+
|
|
29
|
+
self.send_response(200)
|
|
30
|
+
self.send_header("Content-Length", str(len(self.content)))
|
|
31
|
+
self.send_header("Accept-Ranges", "bytes" if self.support_range else "none")
|
|
32
|
+
self.end_headers()
|
|
33
|
+
|
|
34
|
+
def do_GET(self) -> None:
|
|
35
|
+
range_header = self.headers.get("Range")
|
|
36
|
+
self.__class__.requests.append(("GET", range_header))
|
|
37
|
+
if self.fail_all:
|
|
38
|
+
self.send_response(500)
|
|
39
|
+
self.end_headers()
|
|
40
|
+
return
|
|
41
|
+
|
|
42
|
+
if range_header and self.support_range:
|
|
43
|
+
start, end = self._parse_range(range_header)
|
|
44
|
+
if start >= len(self.content):
|
|
45
|
+
self.send_response(416)
|
|
46
|
+
self.send_header("Content-Range", f"bytes */{len(self.content)}")
|
|
47
|
+
self.end_headers()
|
|
48
|
+
return
|
|
49
|
+
|
|
50
|
+
end = min(end, len(self.content) - 1)
|
|
51
|
+
body = self.content[start : end + 1]
|
|
52
|
+
self.send_response(206)
|
|
53
|
+
self.send_header("Content-Length", str(len(body)))
|
|
54
|
+
self.send_header("Content-Range", f"bytes {start}-{end}/{len(self.content)}")
|
|
55
|
+
self.send_header("Accept-Ranges", "bytes")
|
|
56
|
+
self.end_headers()
|
|
57
|
+
self.wfile.write(body)
|
|
58
|
+
return
|
|
59
|
+
|
|
60
|
+
self.send_response(200)
|
|
61
|
+
self.send_header("Content-Length", str(len(self.content)))
|
|
62
|
+
self.send_header("Accept-Ranges", "none")
|
|
63
|
+
self.end_headers()
|
|
64
|
+
self.wfile.write(self.content)
|
|
65
|
+
|
|
66
|
+
@staticmethod
|
|
67
|
+
def _parse_range(header: str) -> tuple[int, int]:
|
|
68
|
+
prefix = "bytes="
|
|
69
|
+
if not header.startswith(prefix):
|
|
70
|
+
return 0, 0
|
|
71
|
+
start_text, end_text = header.removeprefix(prefix).split("-", 1)
|
|
72
|
+
return int(start_text), int(end_text)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@contextlib.contextmanager
|
|
76
|
+
def range_server(*, content: bytes, support_range: bool = True, fail_all: bool = False):
|
|
77
|
+
class Handler(RangeTestHandler):
|
|
78
|
+
pass
|
|
79
|
+
|
|
80
|
+
Handler.content = content
|
|
81
|
+
Handler.support_range = support_range
|
|
82
|
+
Handler.fail_all = fail_all
|
|
83
|
+
Handler.requests = []
|
|
84
|
+
|
|
85
|
+
server = ThreadingHTTPServer(("127.0.0.1", 0), Handler)
|
|
86
|
+
thread = threading.Thread(target=server.serve_forever, daemon=True)
|
|
87
|
+
thread.start()
|
|
88
|
+
try:
|
|
89
|
+
yield f"http://127.0.0.1:{server.server_port}/file.bin", Handler
|
|
90
|
+
finally:
|
|
91
|
+
server.shutdown()
|
|
92
|
+
server.server_close()
|
|
93
|
+
thread.join(timeout=5)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class DownloaderTest(unittest.IsolatedAsyncioTestCase):
|
|
97
|
+
async def test_download_uses_range_parts_when_server_supports_range(self):
|
|
98
|
+
content = bytes(range(251)) * 20
|
|
99
|
+
progresses: list[tuple[int, int]] = []
|
|
100
|
+
|
|
101
|
+
async def progress(current: int, total: int) -> None:
|
|
102
|
+
progresses.append((current, total))
|
|
103
|
+
|
|
104
|
+
with TemporaryDirectory() as tmp, range_server(content=content, support_range=True) as (url, handler):
|
|
105
|
+
target = Path(tmp) / "video.bin"
|
|
106
|
+
|
|
107
|
+
path = await download(
|
|
108
|
+
url,
|
|
109
|
+
target,
|
|
110
|
+
progress=progress,
|
|
111
|
+
connections=4,
|
|
112
|
+
min_split_size=512,
|
|
113
|
+
chunk_size=128,
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
self.assertEqual(Path(path), target)
|
|
117
|
+
self.assertEqual(target.read_bytes(), content)
|
|
118
|
+
self.assertEqual(progresses[-1], (len(content), len(content)))
|
|
119
|
+
range_gets = [range_header for method, range_header in handler.requests if method == "GET" and range_header]
|
|
120
|
+
self.assertGreaterEqual(len(range_gets), 2)
|
|
121
|
+
self.assertFalse(list(Path(tmp).glob(".*.parsehub-tmp")))
|
|
122
|
+
|
|
123
|
+
async def test_download_falls_back_to_single_request_when_range_is_ignored(self):
|
|
124
|
+
content = b"fallback-body" * 100
|
|
125
|
+
|
|
126
|
+
with TemporaryDirectory() as tmp, range_server(content=content, support_range=False) as (url, handler):
|
|
127
|
+
target = Path(tmp) / "image.bin"
|
|
128
|
+
|
|
129
|
+
await download(url, target, connections=4, min_split_size=10, chunk_size=32)
|
|
130
|
+
|
|
131
|
+
self.assertEqual(target.read_bytes(), content)
|
|
132
|
+
self.assertIn(("GET", "bytes=0-0"), handler.requests)
|
|
133
|
+
self.assertIn(("GET", None), handler.requests)
|
|
134
|
+
self.assertFalse(list(Path(tmp).glob(".*.parsehub-tmp")))
|
|
135
|
+
|
|
136
|
+
async def test_download_keeps_existing_file_when_request_fails(self):
|
|
137
|
+
with TemporaryDirectory() as tmp, range_server(content=b"new", support_range=True, fail_all=True) as (url, _):
|
|
138
|
+
target = Path(tmp) / "video.bin"
|
|
139
|
+
target.write_bytes(b"old")
|
|
140
|
+
|
|
141
|
+
with self.assertRaises(DownloadError):
|
|
142
|
+
await download(url, target, connections=4, max_retries=0)
|
|
143
|
+
|
|
144
|
+
self.assertEqual(target.read_bytes(), b"old")
|
|
145
|
+
self.assertFalse(list(Path(tmp).glob(".*.parsehub-tmp")))
|
|
146
|
+
|
|
147
|
+
async def test_connections_one_uses_single_request(self):
|
|
148
|
+
content = b"single" * 200
|
|
149
|
+
|
|
150
|
+
with TemporaryDirectory() as tmp, range_server(content=content, support_range=True) as (url, handler):
|
|
151
|
+
target = Path(tmp) / "single.bin"
|
|
152
|
+
|
|
153
|
+
await download(url, target, connections=1, min_split_size=10)
|
|
154
|
+
|
|
155
|
+
self.assertEqual(target.read_bytes(), content)
|
|
156
|
+
range_gets = [range_header for method, range_header in handler.requests if method == "GET" and range_header]
|
|
157
|
+
self.assertEqual(range_gets, [])
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
if __name__ == "__main__":
|
|
161
|
+
unittest.main()
|
|
@@ -1,79 +0,0 @@
|
|
|
1
|
-
import re
|
|
2
|
-
from collections.abc import Iterator
|
|
3
|
-
from typing import Any, NamedTuple
|
|
4
|
-
|
|
5
|
-
import requests
|
|
6
|
-
from instaloader import InstaloaderContext, InstaloaderException, Post
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
class MyPostSidecarNode(NamedTuple):
|
|
10
|
-
is_video: bool
|
|
11
|
-
display_url: str
|
|
12
|
-
video_url: str | None
|
|
13
|
-
width: int
|
|
14
|
-
height: int
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
class MyPost(Post):
|
|
18
|
-
def get_sidecar_nodes(self, start: int = 0, end: int = -1) -> Iterator[MyPostSidecarNode]: # type: ignore[override]
|
|
19
|
-
if self.typename == "GraphSidecar":
|
|
20
|
-
edges = self._field("edge_sidecar_to_children", "edges")
|
|
21
|
-
if end < 0:
|
|
22
|
-
end = len(edges) - 1
|
|
23
|
-
if start < 0:
|
|
24
|
-
start = len(edges) - 1
|
|
25
|
-
if any(edge["node"]["is_video"] and "video_url" not in edge["node"] for edge in edges[start : (end + 1)]):
|
|
26
|
-
# video_url is only present in full metadata, issue #558.
|
|
27
|
-
edges = self._full_metadata["edge_sidecar_to_children"]["edges"]
|
|
28
|
-
for idx, edge in enumerate(edges):
|
|
29
|
-
if start <= idx <= end:
|
|
30
|
-
node = edge["node"]
|
|
31
|
-
is_video = node["is_video"]
|
|
32
|
-
display_url = node["display_url"]
|
|
33
|
-
dimensions = node["dimensions"]
|
|
34
|
-
width = dimensions["width"]
|
|
35
|
-
height = dimensions["height"]
|
|
36
|
-
|
|
37
|
-
if not is_video and self._context.iphone_support and self._context.is_logged_in:
|
|
38
|
-
try:
|
|
39
|
-
carousel_media = self._iphone_struct["carousel_media"]
|
|
40
|
-
orig_url = carousel_media[idx]["image_versions2"]["candidates"][0]["url"]
|
|
41
|
-
display_url = re.sub(r"([?&])se=\d+&?", r"\1", orig_url).rstrip("&")
|
|
42
|
-
except (InstaloaderException, KeyError, IndexError) as err:
|
|
43
|
-
self._context.error(f"Unable to fetch high quality image version of {self}: {err}")
|
|
44
|
-
yield MyPostSidecarNode(
|
|
45
|
-
is_video=is_video,
|
|
46
|
-
display_url=display_url,
|
|
47
|
-
video_url=node["video_url"] if is_video else None,
|
|
48
|
-
width=width,
|
|
49
|
-
height=height,
|
|
50
|
-
)
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
class MyInstaloaderContext(InstaloaderContext):
|
|
54
|
-
"""
|
|
55
|
-
支持自定义代理
|
|
56
|
-
"""
|
|
57
|
-
|
|
58
|
-
def __init__(self, proxy: str | None = None, cookie: dict | None = None):
|
|
59
|
-
self.proxy: dict[str, str | None] = {"http": proxy, "https": proxy}
|
|
60
|
-
self.cookie = cookie
|
|
61
|
-
super().__init__()
|
|
62
|
-
|
|
63
|
-
def get_anonymous_session(self) -> requests.Session:
|
|
64
|
-
session = super().get_anonymous_session()
|
|
65
|
-
if self.proxy:
|
|
66
|
-
session.proxies = {k: v for k, v in self.proxy.items() if v is not None}
|
|
67
|
-
session.trust_env = False
|
|
68
|
-
return session
|
|
69
|
-
|
|
70
|
-
def get_json(self, *args: Any, **kwargs: Any) -> Any:
|
|
71
|
-
session = kwargs.get("session")
|
|
72
|
-
if isinstance(session, requests.Session):
|
|
73
|
-
if self.proxy:
|
|
74
|
-
session.proxies = {k: v for k, v in self.proxy.items() if v is not None}
|
|
75
|
-
session.trust_env = False
|
|
76
|
-
if self.cookie:
|
|
77
|
-
session.cookies.update(self.cookie)
|
|
78
|
-
|
|
79
|
-
return super().get_json(*args, **kwargs)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|