parsehub 2.0.32__tar.gz → 2.0.34__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. {parsehub-2.0.32/src/parsehub.egg-info → parsehub-2.0.34}/PKG-INFO +1 -2
  2. {parsehub-2.0.32 → parsehub-2.0.34}/pyproject.toml +1 -2
  3. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/instagram.py +9 -27
  4. parsehub-2.0.34/src/parsehub/provider_api/instagram.py +380 -0
  5. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/xhs.py +17 -2
  6. {parsehub-2.0.32 → parsehub-2.0.34/src/parsehub.egg-info}/PKG-INFO +1 -2
  7. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub.egg-info/requires.txt +0 -1
  8. parsehub-2.0.32/src/parsehub/provider_api/instagram.py +0 -79
  9. {parsehub-2.0.32 → parsehub-2.0.34}/LICENSE +0 -0
  10. {parsehub-2.0.32 → parsehub-2.0.34}/README.md +0 -0
  11. {parsehub-2.0.32 → parsehub-2.0.34}/setup.cfg +0 -0
  12. {parsehub-2.0.32 → parsehub-2.0.34}/src/__init__.py +0 -0
  13. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/__init__.py +0 -0
  14. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/cli.py +0 -0
  15. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/cli_config.py +0 -0
  16. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/config/__init__.py +0 -0
  17. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/config/config.py +0 -0
  18. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/errors.py +0 -0
  19. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/__init__.py +0 -0
  20. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/base/__init__.py +0 -0
  21. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/base/base.py +0 -0
  22. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/base/ytdlp.py +0 -0
  23. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/__init__.py +0 -0
  24. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/bilibili.py +0 -0
  25. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/coolapk.py +0 -0
  26. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/douyin.py +0 -0
  27. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/facebook.py +0 -0
  28. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/kuaishou.py +0 -0
  29. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/pipix.py +0 -0
  30. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/snapchat.py +0 -0
  31. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/threads.py +0 -0
  32. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/tieba.py +0 -0
  33. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/tiktok.py +0 -0
  34. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/twitter.py +0 -0
  35. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/weibo.py +0 -0
  36. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/weixin.py +0 -0
  37. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/xhs.py +0 -0
  38. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/xiaoheihe.py +0 -0
  39. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/youtube.py +0 -0
  40. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/parsers/parser/zuiyou.py +0 -0
  41. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/__init__.py +0 -0
  42. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/bilibili.py +0 -0
  43. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/coolapk.py +0 -0
  44. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/douyin.py +0 -0
  45. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/kuaishou.py +0 -0
  46. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/pipix.py +0 -0
  47. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/threads.py +0 -0
  48. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/tieba.py +0 -0
  49. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/tiktok.py +0 -0
  50. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/twitter.py +0 -0
  51. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/weibo.py +0 -0
  52. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/weixin.py +0 -0
  53. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/xiaoheihe.py +0 -0
  54. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/provider_api/zuiyou.py +0 -0
  55. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/types/__init__.py +0 -0
  56. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/types/callback.py +0 -0
  57. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/types/media_file.py +0 -0
  58. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/types/media_ref.py +0 -0
  59. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/types/platform.py +0 -0
  60. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/types/post.py +0 -0
  61. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/types/result.py +0 -0
  62. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/utils/downloader.py +0 -0
  63. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/utils/helpers.py +0 -0
  64. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub/utils/media_info.py +0 -0
  65. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub.egg-info/SOURCES.txt +0 -0
  66. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub.egg-info/dependency_links.txt +0 -0
  67. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub.egg-info/entry_points.txt +0 -0
  68. {parsehub-2.0.32 → parsehub-2.0.34}/src/parsehub.egg-info/top_level.txt +0 -0
  69. {parsehub-2.0.32 → parsehub-2.0.34}/test/test_cli.py +0 -0
  70. {parsehub-2.0.32 → parsehub-2.0.34}/test/test_cli_config.py +0 -0
  71. {parsehub-2.0.32 → parsehub-2.0.34}/test/test_core_offline.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: parsehub
3
- Version: 2.0.32
3
+ Version: 2.0.34
4
4
  Summary: 轻量、异步、开箱即用的社交媒体聚合解析库
5
5
  Author-email: 梓澪 <zilingmio@gmail.com>
6
6
  License: MIT
@@ -24,7 +24,6 @@ Requires-Dist: tenacity>=8.5.0
24
24
  Requires-Dist: urlextract>=1.9.0
25
25
  Requires-Dist: yt-dlp[default]
26
26
  Requires-Dist: lxml>=5.3.0
27
- Requires-Dist: instaloader>=4.14
28
27
  Requires-Dist: pydantic>=1.10.19
29
28
  Requires-Dist: markdownify>=1.1.0
30
29
  Requires-Dist: markdown>=3.7
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "parsehub"
3
- version = "2.0.32"
3
+ version = "2.0.34"
4
4
  description = "轻量、异步、开箱即用的社交媒体聚合解析库"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.12.0"
@@ -27,7 +27,6 @@ dependencies = [
27
27
  "urlextract>=1.9.0",
28
28
  "yt-dlp[default]",
29
29
  "lxml>=5.3.0",
30
- "instaloader>=4.14",
31
30
  "pydantic>=1.10.19",
32
31
  "markdownify>=1.1.0",
33
32
  "markdown>=3.7",
@@ -1,10 +1,6 @@
1
- import asyncio
2
1
  import re
3
- from typing import cast
4
2
 
5
- from instaloader import BadResponseException
6
-
7
- from ...provider_api.instagram import MyInstaloaderContext, MyPost
3
+ from ...provider_api.instagram import InstagramAPI, InstagramAPIError, InstagramMediaType, InstagramPost
8
4
  from ...types import ImageParseResult, ImageRef, MultimediaParseResult, ParseError, Platform, VideoParseResult, VideoRef
9
5
  from ...utils.helpers import SecretCookie
10
6
  from ..base.base import BaseParser
@@ -23,14 +19,10 @@ class InstagramParser(BaseParser):
23
19
 
24
20
  post = await self._parse(raw_url, shortcode)
25
21
 
26
- try:
27
- dimensions: dict = post._field("dimensions")
28
- except KeyError:
29
- dimensions = {}
30
- width, height = dimensions.get("width", 0) or 0, dimensions.get("height", 0) or 0
22
+ width, height = post.width, post.height
31
23
 
32
24
  match post.typename:
33
- case "GraphSidecar":
25
+ case InstagramMediaType.SIDECAR:
34
26
  media = [
35
27
  VideoRef(url=i.video_url, thumb_url=i.display_url, width=i.width, height=i.height)
36
28
  if i.is_video and i.video_url
@@ -38,11 +30,11 @@ class InstagramParser(BaseParser):
38
30
  for i in post.get_sidecar_nodes()
39
31
  ]
40
32
  return MultimediaParseResult(media=media, title=post.title, content=post.caption)
41
- case "GraphImage":
33
+ case InstagramMediaType.IMAGE:
42
34
  return ImageParseResult(
43
35
  photo=[ImageRef(url=post.url, width=width, height=height)], title=post.title, content=post.caption
44
36
  )
45
- case "GraphVideo":
37
+ case InstagramMediaType.VIDEO:
46
38
  return VideoParseResult(
47
39
  video=VideoRef(
48
40
  url=post.video_url or post.url,
@@ -57,19 +49,11 @@ class InstagramParser(BaseParser):
57
49
  case _:
58
50
  raise ParseError("不支持的类型")
59
51
 
60
- async def _parse(self, url: str, shortcode: str, cookie: SecretCookie | None = None) -> MyPost:
52
+ async def _parse(self, url: str, shortcode: str, cookie: SecretCookie | None = None) -> InstagramPost:
61
53
  try:
62
- post = await asyncio.wait_for(
63
- asyncio.to_thread(
64
- MyPost.from_shortcode,
65
- MyInstaloaderContext(self.proxy, cookie.get_value() if cookie else None),
66
- shortcode,
67
- ),
68
- 30,
69
- )
70
- except TimeoutError as e:
71
- raise ParseError("解析超时") from e
72
- except BadResponseException as e:
54
+ api = InstagramAPI(proxy=self.proxy, cookie=cookie.get_value() if cookie else None, timeout=30)
55
+ return await api.get_post(shortcode)
56
+ except InstagramAPIError as e:
73
57
  match str(e):
74
58
  case "Fetching Post metadata failed.":
75
59
  if self.cookie and cookie is None:
@@ -84,8 +68,6 @@ class InstagramParser(BaseParser):
84
68
  else:
85
69
  text = str(e)
86
70
  raise ParseError(f"无法获取帖子内容: {text}") from e
87
- else:
88
- return cast(MyPost, post)
89
71
 
90
72
  @staticmethod
91
73
  def get_short_code(url: str) -> str | None:
@@ -0,0 +1,380 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from collections.abc import Iterator
5
+ from dataclasses import dataclass
6
+ from enum import StrEnum
7
+ from typing import Any, cast
8
+
9
+ import httpx
10
+
11
+
12
+ class InstagramAPIError(RuntimeError):
13
+ """Instagram 接口请求或响应解析失败。"""
14
+
15
+
16
+ class InstagramMediaType(StrEnum):
17
+ IMAGE = "GraphImage"
18
+ VIDEO = "GraphVideo"
19
+ SIDECAR = "GraphSidecar"
20
+
21
+
22
+ @dataclass(slots=True)
23
+ class InstagramSidecarNode:
24
+ is_video: bool
25
+ display_url: str
26
+ video_url: str | None
27
+ width: int
28
+ height: int
29
+
30
+
31
+ class InstagramPost:
32
+ """轻量版 Instagram Post,只保留当前项目解析帖子需要的字段。"""
33
+
34
+ _XDT_TYPES = {
35
+ "XDTGraphImage": InstagramMediaType.IMAGE,
36
+ "XDTGraphVideo": InstagramMediaType.VIDEO,
37
+ "XDTGraphSidecar": InstagramMediaType.SIDECAR,
38
+ }
39
+
40
+ def __init__(self, node: dict[str, Any]):
41
+ self._node = node
42
+ self._normalize_typename()
43
+
44
+ def _normalize_typename(self) -> None:
45
+ typename = self._node.get("__typename")
46
+ if typename in self._XDT_TYPES:
47
+ self._node["__typename"] = self._XDT_TYPES[typename]
48
+
49
+ def _field(self, *keys: str) -> Any:
50
+ value: Any = self._node
51
+ for key in keys:
52
+ value = value[key]
53
+ return value
54
+
55
+ @property
56
+ def shortcode(self) -> str:
57
+ return str(self._node.get("shortcode") or self._node["code"])
58
+
59
+ @property
60
+ def typename(self) -> InstagramMediaType:
61
+ return InstagramMediaType(self._field("__typename"))
62
+
63
+ @property
64
+ def is_video(self) -> bool:
65
+ return bool(self._field("is_video"))
66
+
67
+ @property
68
+ def title(self) -> str | None:
69
+ return self._node.get("title")
70
+
71
+ @property
72
+ def caption(self) -> str | None:
73
+ caption_edges = self._node.get("edge_media_to_caption", {}).get("edges") or []
74
+ if caption_edges:
75
+ if text := caption_edges[0].get("node", {}).get("text"):
76
+ return str(text)
77
+ return None
78
+ return self._node.get("caption")
79
+
80
+ @property
81
+ def url(self) -> str:
82
+ return str(self._node.get("display_url") or self._node["display_src"])
83
+
84
+ @property
85
+ def video_url(self) -> str | None:
86
+ if not self.is_video:
87
+ return None
88
+ return self._node.get("video_url")
89
+
90
+ @property
91
+ def video_duration(self) -> float | None:
92
+ value = self._node.get("video_duration")
93
+ return float(value) if value is not None else None
94
+
95
+ @property
96
+ def width(self) -> int:
97
+ return int(self._node.get("dimensions", {}).get("width") or 0)
98
+
99
+ @property
100
+ def height(self) -> int:
101
+ return int(self._node.get("dimensions", {}).get("height") or 0)
102
+
103
+ def get_sidecar_nodes(self, start: int = 0, end: int = -1) -> Iterator[InstagramSidecarNode]:
104
+ if self.typename is not InstagramMediaType.SIDECAR:
105
+ return
106
+
107
+ edges = self._field("edge_sidecar_to_children", "edges")
108
+ if end < 0:
109
+ end = len(edges) - 1
110
+ if start < 0:
111
+ start = len(edges) - 1
112
+
113
+ for idx, edge in enumerate(edges):
114
+ if not start <= idx <= end:
115
+ continue
116
+
117
+ node = edge["node"]
118
+ dimensions = node.get("dimensions", {})
119
+ is_video = bool(node.get("is_video"))
120
+ yield InstagramSidecarNode(
121
+ is_video=is_video,
122
+ display_url=node.get("display_url") or node.get("display_src") or "",
123
+ video_url=node.get("video_url") if is_video else None,
124
+ width=int(dimensions.get("width") or 0),
125
+ height=int(dimensions.get("height") or 0),
126
+ )
127
+
128
+
129
+ class InstagramAPI:
130
+ GRAPHQL_URL = "https://www.instagram.com/graphql/query"
131
+ INSTAGRAM_URL = "https://www.instagram.com/"
132
+ SHORTCODE_DOC_ID = "27128499623469141"
133
+
134
+ DEFAULT_COOKIES = {
135
+ "sessionid": "",
136
+ "mid": "",
137
+ "ig_pr": "1",
138
+ "ig_vw": "1920",
139
+ "csrftoken": "",
140
+ "s_network": "",
141
+ "ds_user_id": "",
142
+ }
143
+
144
+ DEFAULT_USER_AGENT = (
145
+ "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
146
+ "AppleWebKit/537.36 (KHTML, like Gecko) "
147
+ "Chrome/124.0.0.0 Safari/537.36"
148
+ )
149
+
150
+ MEDIA_TYPES = {
151
+ 1: InstagramMediaType.IMAGE,
152
+ 2: InstagramMediaType.VIDEO,
153
+ 8: InstagramMediaType.SIDECAR,
154
+ }
155
+
156
+ def __init__(
157
+ self,
158
+ *,
159
+ proxy: str | None = None,
160
+ cookie: dict[str, str] | None = None,
161
+ timeout: float = 30,
162
+ user_agent: str | None = None,
163
+ ):
164
+ self.proxy = proxy
165
+ self.cookie = cookie or {}
166
+ self.timeout = timeout
167
+ self.user_agent = user_agent or self.DEFAULT_USER_AGENT
168
+
169
+ async def get_post(self, shortcode: str) -> InstagramPost:
170
+ media = await self.get_shortcode_media(shortcode)
171
+ return InstagramPost(media)
172
+
173
+ async def get_shortcode_media(self, shortcode: str) -> dict[str, Any]:
174
+ payload = await self._post_graphql(
175
+ doc_id=self.SHORTCODE_DOC_ID,
176
+ variables={
177
+ "shortcode": shortcode,
178
+ "__relay_internal__pv__PolarisAIGMMediaWebLabelEnabledrelayprovider": False,
179
+ },
180
+ )
181
+
182
+ media = self._extract_shortcode_media(payload)
183
+ if media is None:
184
+ raise InstagramAPIError("Fetching Post metadata failed.")
185
+ return media
186
+
187
+ async def _post_graphql(self, *, doc_id: str, variables: dict[str, Any]) -> dict[str, Any]:
188
+ async with self._new_client() as client:
189
+ await self._ensure_csrf_token(client)
190
+ data = {
191
+ "variables": json.dumps(variables, separators=(",", ":")),
192
+ "doc_id": doc_id,
193
+ "server_timestamps": "true",
194
+ }
195
+
196
+ try:
197
+ response = await client.post(self.GRAPHQL_URL, data=data, follow_redirects=False)
198
+ except httpx.HTTPError as exc:
199
+ raise InstagramAPIError(f"请求 Instagram GraphQL 失败: {exc}") from exc
200
+
201
+ if response.status_code != 200:
202
+ raise InstagramAPIError(f"Instagram GraphQL 返回 HTTP {response.status_code}: {response.text[:500]}")
203
+
204
+ try:
205
+ payload = response.json()
206
+ except ValueError as exc:
207
+ raise InstagramAPIError(f"Instagram GraphQL 返回非 JSON 响应: {response.text[:500]}") from exc
208
+
209
+ if payload.get("status") not in (None, "ok"):
210
+ raise InstagramAPIError(f"Instagram GraphQL 状态异常: {payload!r}")
211
+ return cast(dict[str, Any], payload)
212
+
213
+ def _extract_shortcode_media(self, payload: dict[str, Any]) -> dict[str, Any] | None:
214
+ data = payload.get("data")
215
+ if not isinstance(data, dict):
216
+ raise InstagramAPIError(f"响应缺少 data: {payload!r}")
217
+
218
+ old_media = data.get("xdt_shortcode_media")
219
+ if isinstance(old_media, dict):
220
+ return old_media
221
+ if old_media is None and "xdt_shortcode_media" in data:
222
+ return None
223
+
224
+ web_info = data.get("xdt_api__v1__media__shortcode__web_info")
225
+ if not isinstance(web_info, dict):
226
+ raise InstagramAPIError(f"响应缺少 shortcode web_info: {payload!r}")
227
+
228
+ items = web_info.get("items") or []
229
+ if not items:
230
+ return None
231
+ if not isinstance(items[0], dict):
232
+ raise InstagramAPIError(f"shortcode web_info.items[0] 类型异常: {type(items[0]).__name__}")
233
+ return self._convert_v1_media(items[0])
234
+
235
+ def _convert_v1_media(self, media: dict[str, Any]) -> dict[str, Any]:
236
+ media_type = self._media_type(media)
237
+ typename = self.MEDIA_TYPES.get(media_type, InstagramMediaType.IMAGE)
238
+ caption = media.get("caption")
239
+ caption_text = caption.get("text") if isinstance(caption, dict) else caption
240
+ image_candidate = self._first_image_candidate(media)
241
+ video_version = self._first_video_version(media)
242
+
243
+ node: dict[str, Any] = {
244
+ "shortcode": media.get("code", ""),
245
+ "id": media.get("pk", ""),
246
+ "__typename": typename,
247
+ "is_video": media_type == 2,
248
+ "taken_at_timestamp": media.get("taken_at"),
249
+ "edge_media_to_caption": {
250
+ "edges": [{"node": {"text": caption_text}}] if caption_text else [],
251
+ },
252
+ "edge_media_preview_like": {"count": media.get("like_count") or 0},
253
+ "edge_media_to_parent_comment": {
254
+ "count": media.get("comment_count") or 0,
255
+ "edges": [],
256
+ },
257
+ "owner": self._convert_owner(media.get("user")),
258
+ "dimensions": self._dimensions_from_candidate(image_candidate),
259
+ "display_url": image_candidate.get("url", ""),
260
+ }
261
+
262
+ for source_key, target_key in (
263
+ ("title", "title"),
264
+ ("has_liked", "viewer_has_liked"),
265
+ ("accessibility_caption", "accessibility_caption"),
266
+ ("location", "location"),
267
+ ("video_duration", "video_duration"),
268
+ ("view_count", "video_view_count"),
269
+ ("play_count", "video_play_count"),
270
+ ):
271
+ if media.get(source_key) is not None:
272
+ node[target_key] = media[source_key]
273
+
274
+ if video_version.get("url"):
275
+ node["video_url"] = video_version["url"]
276
+
277
+ carousel_media = media.get("carousel_media") or []
278
+ if carousel_media:
279
+ node["edge_sidecar_to_children"] = {
280
+ "edges": [{"node": self._convert_v1_sidecar_item(item)} for item in carousel_media],
281
+ }
282
+
283
+ return node
284
+
285
+ def _convert_v1_sidecar_item(self, item: dict[str, Any]) -> dict[str, Any]:
286
+ media_type = self._media_type(item)
287
+ typename = self.MEDIA_TYPES.get(media_type, InstagramMediaType.IMAGE)
288
+ image_candidate = self._first_image_candidate(item)
289
+ video_version = self._first_video_version(item)
290
+ is_video = media_type == 2
291
+
292
+ node: dict[str, Any] = {
293
+ "shortcode": item.get("code", ""),
294
+ "__typename": typename,
295
+ "is_video": is_video,
296
+ "display_url": image_candidate.get("url", ""),
297
+ "video_url": video_version.get("url") if is_video else None,
298
+ "dimensions": self._dimensions_from_candidate(image_candidate),
299
+ }
300
+ if item.get("accessibility_caption") is not None:
301
+ node["accessibility_caption"] = item["accessibility_caption"]
302
+ return node
303
+
304
+ @staticmethod
305
+ def _media_type(media: dict[str, Any]) -> int:
306
+ value = media.get("media_type")
307
+ return value if isinstance(value, int) else 0
308
+
309
+ @staticmethod
310
+ def _convert_owner(user: Any) -> dict[str, Any]:
311
+ if not isinstance(user, dict):
312
+ return {"id": "", "username": "", "full_name": ""}
313
+ return {
314
+ "id": user.get("pk", ""),
315
+ "username": user.get("username", ""),
316
+ "full_name": user.get("full_name", ""),
317
+ }
318
+
319
+ @staticmethod
320
+ def _first_image_candidate(media: dict[str, Any]) -> dict[str, Any]:
321
+ candidates = media.get("image_versions2", {}).get("candidates") or []
322
+ return candidates[0] if candidates and isinstance(candidates[0], dict) else {}
323
+
324
+ @staticmethod
325
+ def _first_video_version(media: dict[str, Any]) -> dict[str, Any]:
326
+ versions = media.get("video_versions") or []
327
+ return versions[0] if versions and isinstance(versions[0], dict) else {}
328
+
329
+ @staticmethod
330
+ def _dimensions_from_candidate(candidate: dict[str, Any]) -> dict[str, int]:
331
+ return {
332
+ "width": int(candidate.get("width") or 0),
333
+ "height": int(candidate.get("height") or 0),
334
+ }
335
+
336
+ async def _ensure_csrf_token(self, client: httpx.AsyncClient) -> None:
337
+ csrf_token = self._get_cookie_value(client, "csrftoken")
338
+ if not csrf_token:
339
+ try:
340
+ await client.get(self.INSTAGRAM_URL, follow_redirects=True)
341
+ except httpx.HTTPError as exc:
342
+ raise InstagramAPIError(f"获取 Instagram csrftoken 失败: {exc}") from exc
343
+ csrf_token = self._get_cookie_value(client, "csrftoken")
344
+
345
+ if not csrf_token:
346
+ raise InstagramAPIError("无法获取 Instagram csrftoken")
347
+ client.headers["x-csrftoken"] = csrf_token
348
+
349
+ @staticmethod
350
+ def _get_cookie_value(client: httpx.AsyncClient, name: str) -> str:
351
+ values = [cookie.value for cookie in client.cookies.jar if cookie.name == name and cookie.value]
352
+ return values[-1] if values else ""
353
+
354
+ def _new_client(self) -> httpx.AsyncClient:
355
+ cookies = self.DEFAULT_COOKIES | self.cookie
356
+ headers = {
357
+ "Accept": "*/*",
358
+ "Accept-Encoding": "gzip, deflate",
359
+ "Accept-Language": "en-US,en;q=0.8",
360
+ "Content-Type": "application/x-www-form-urlencoded",
361
+ "Referer": "https://www.instagram.com/",
362
+ "User-Agent": self.user_agent,
363
+ "authority": "www.instagram.com",
364
+ "scheme": "https",
365
+ }
366
+ return httpx.AsyncClient(
367
+ cookies=cookies,
368
+ headers=headers,
369
+ proxy=self.proxy,
370
+ timeout=self.timeout,
371
+ )
372
+
373
+
374
+ __all__ = [
375
+ "InstagramAPI",
376
+ "InstagramAPIError",
377
+ "InstagramMediaType",
378
+ "InstagramPost",
379
+ "InstagramSidecarNode",
380
+ ]
@@ -100,7 +100,7 @@ class XHSAPI:
100
100
  stream = selected_stream[0]
101
101
  image = XHSMedia(
102
102
  XHSMediaType.LIVE_PHOTO,
103
- thumb_url=i["urlDefault"],
103
+ thumb_url=self.get_raw_image_url(i["urlDefault"]),
104
104
  url=stream["masterUrl"],
105
105
  width=i["width"],
106
106
  height=i["height"],
@@ -108,7 +108,7 @@ class XHSAPI:
108
108
  else:
109
109
  image = XHSMedia(
110
110
  XHSMediaType.IMAGE,
111
- url=i["urlDefault"],
111
+ url=self.get_raw_image_url(i["urlDefault"]),
112
112
  thumb_url=i["urlPre"],
113
113
  width=i["width"],
114
114
  height=i["height"],
@@ -120,6 +120,21 @@ class XHSAPI:
120
120
  html = await self.__fetch_html(url)
121
121
  return self.__parse(await self.__extract_data(html))
122
122
 
123
+ @staticmethod
124
+ def get_trace_id(img_url: str) -> str:
125
+ trace_id = img_url.split("/")[-1].split("!")[0]
126
+ if "spectrum" in img_url:
127
+ return "spectrum/" + trace_id
128
+ if "note_pre_post_uhdr" in img_url:
129
+ return "note_pre_post_uhdr/" + trace_id
130
+ if "notes_pre_post" in img_url:
131
+ return "notes_pre_post/" + trace_id
132
+ return trace_id
133
+
134
+ def get_raw_image_url(self, ime_url: str) -> str:
135
+ """拼接无水印图片链接"""
136
+ return f"http://sns-img-hw.xhscdn.com/{self.get_trace_id(ime_url)}"
137
+
123
138
 
124
139
  class XHSMediaType(Enum):
125
140
  IMAGE = "image"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: parsehub
3
- Version: 2.0.32
3
+ Version: 2.0.34
4
4
  Summary: 轻量、异步、开箱即用的社交媒体聚合解析库
5
5
  Author-email: 梓澪 <zilingmio@gmail.com>
6
6
  License: MIT
@@ -24,7 +24,6 @@ Requires-Dist: tenacity>=8.5.0
24
24
  Requires-Dist: urlextract>=1.9.0
25
25
  Requires-Dist: yt-dlp[default]
26
26
  Requires-Dist: lxml>=5.3.0
27
- Requires-Dist: instaloader>=4.14
28
27
  Requires-Dist: pydantic>=1.10.19
29
28
  Requires-Dist: markdownify>=1.1.0
30
29
  Requires-Dist: markdown>=3.7
@@ -7,7 +7,6 @@ tenacity>=8.5.0
7
7
  urlextract>=1.9.0
8
8
  yt-dlp[default]
9
9
  lxml>=5.3.0
10
- instaloader>=4.14
11
10
  pydantic>=1.10.19
12
11
  markdownify>=1.1.0
13
12
  markdown>=3.7
@@ -1,79 +0,0 @@
1
- import re
2
- from collections.abc import Iterator
3
- from typing import Any, NamedTuple
4
-
5
- import requests
6
- from instaloader import InstaloaderContext, InstaloaderException, Post
7
-
8
-
9
- class MyPostSidecarNode(NamedTuple):
10
- is_video: bool
11
- display_url: str
12
- video_url: str | None
13
- width: int
14
- height: int
15
-
16
-
17
- class MyPost(Post):
18
- def get_sidecar_nodes(self, start: int = 0, end: int = -1) -> Iterator[MyPostSidecarNode]: # type: ignore[override]
19
- if self.typename == "GraphSidecar":
20
- edges = self._field("edge_sidecar_to_children", "edges")
21
- if end < 0:
22
- end = len(edges) - 1
23
- if start < 0:
24
- start = len(edges) - 1
25
- if any(edge["node"]["is_video"] and "video_url" not in edge["node"] for edge in edges[start : (end + 1)]):
26
- # video_url is only present in full metadata, issue #558.
27
- edges = self._full_metadata["edge_sidecar_to_children"]["edges"]
28
- for idx, edge in enumerate(edges):
29
- if start <= idx <= end:
30
- node = edge["node"]
31
- is_video = node["is_video"]
32
- display_url = node["display_url"]
33
- dimensions = node["dimensions"]
34
- width = dimensions["width"]
35
- height = dimensions["height"]
36
-
37
- if not is_video and self._context.iphone_support and self._context.is_logged_in:
38
- try:
39
- carousel_media = self._iphone_struct["carousel_media"]
40
- orig_url = carousel_media[idx]["image_versions2"]["candidates"][0]["url"]
41
- display_url = re.sub(r"([?&])se=\d+&?", r"\1", orig_url).rstrip("&")
42
- except (InstaloaderException, KeyError, IndexError) as err:
43
- self._context.error(f"Unable to fetch high quality image version of {self}: {err}")
44
- yield MyPostSidecarNode(
45
- is_video=is_video,
46
- display_url=display_url,
47
- video_url=node["video_url"] if is_video else None,
48
- width=width,
49
- height=height,
50
- )
51
-
52
-
53
- class MyInstaloaderContext(InstaloaderContext):
54
- """
55
- 支持自定义代理
56
- """
57
-
58
- def __init__(self, proxy: str | None = None, cookie: dict | None = None):
59
- self.proxy: dict[str, str | None] = {"http": proxy, "https": proxy}
60
- self.cookie = cookie
61
- super().__init__()
62
-
63
- def get_anonymous_session(self) -> requests.Session:
64
- session = super().get_anonymous_session()
65
- if self.proxy:
66
- session.proxies = {k: v for k, v in self.proxy.items() if v is not None}
67
- session.trust_env = False
68
- return session
69
-
70
- def get_json(self, *args: Any, **kwargs: Any) -> Any:
71
- session = kwargs.get("session")
72
- if isinstance(session, requests.Session):
73
- if self.proxy:
74
- session.proxies = {k: v for k, v in self.proxy.items() if v is not None}
75
- session.trust_env = False
76
- if self.cookie:
77
- session.cookies.update(self.cookie)
78
-
79
- return super().get_json(*args, **kwargs)
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes