parsehub 2.1.0__tar.gz → 2.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. {parsehub-2.1.0/src/parsehub.egg-info → parsehub-2.1.1}/PKG-INFO +24 -21
  2. {parsehub-2.1.0 → parsehub-2.1.1}/README.md +23 -20
  3. {parsehub-2.1.0 → parsehub-2.1.1}/pyproject.toml +1 -1
  4. parsehub-2.1.1/src/parsehub/parsers/parser/zhihu.py +64 -0
  5. parsehub-2.1.1/src/parsehub/provider_api/zhihu.py +748 -0
  6. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/types/platform.py +1 -0
  7. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/types/result.py +1 -1
  8. {parsehub-2.1.0 → parsehub-2.1.1/src/parsehub.egg-info}/PKG-INFO +24 -21
  9. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub.egg-info/SOURCES.txt +2 -0
  10. {parsehub-2.1.0 → parsehub-2.1.1}/test/test_core_offline.py +6 -0
  11. {parsehub-2.1.0 → parsehub-2.1.1}/LICENSE +0 -0
  12. {parsehub-2.1.0 → parsehub-2.1.1}/setup.cfg +0 -0
  13. {parsehub-2.1.0 → parsehub-2.1.1}/src/__init__.py +0 -0
  14. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/__init__.py +0 -0
  15. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/cli.py +0 -0
  16. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/cli_config.py +0 -0
  17. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/config/__init__.py +0 -0
  18. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/config/config.py +0 -0
  19. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/errors.py +0 -0
  20. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/__init__.py +0 -0
  21. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/base/__init__.py +0 -0
  22. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/base/base.py +0 -0
  23. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/base/ytdlp.py +0 -0
  24. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/__init__.py +0 -0
  25. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/bilibili.py +0 -0
  26. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/coolapk.py +0 -0
  27. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/douyin.py +0 -0
  28. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/facebook.py +0 -0
  29. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/instagram.py +0 -0
  30. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/kuaishou.py +0 -0
  31. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/pipix.py +0 -0
  32. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/snapchat.py +0 -0
  33. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/threads.py +0 -0
  34. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/tieba.py +0 -0
  35. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/tiktok.py +0 -0
  36. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/twitter.py +0 -0
  37. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/weibo.py +0 -0
  38. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/weixin.py +0 -0
  39. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/xhs.py +0 -0
  40. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/xiaoheihe.py +0 -0
  41. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/youtube.py +0 -0
  42. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/parsers/parser/zuiyou.py +0 -0
  43. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/provider_api/__init__.py +0 -0
  44. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/provider_api/bilibili.py +0 -0
  45. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/provider_api/coolapk.py +0 -0
  46. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/provider_api/douyin.py +0 -0
  47. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/provider_api/instagram.py +0 -0
  48. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/provider_api/kuaishou.py +0 -0
  49. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/provider_api/pipix.py +0 -0
  50. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/provider_api/threads.py +0 -0
  51. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/provider_api/tieba.py +0 -0
  52. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/provider_api/tiktok.py +0 -0
  53. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/provider_api/twitter.py +0 -0
  54. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/provider_api/weibo.py +0 -0
  55. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/provider_api/weixin.py +0 -0
  56. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/provider_api/xhs.py +0 -0
  57. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/provider_api/xiaoheihe.py +0 -0
  58. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/provider_api/zuiyou.py +0 -0
  59. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/types/__init__.py +0 -0
  60. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/types/callback.py +0 -0
  61. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/types/media_file.py +0 -0
  62. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/types/media_ref.py +0 -0
  63. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/types/post.py +0 -0
  64. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/utils/downloader.py +0 -0
  65. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/utils/helpers.py +0 -0
  66. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub/utils/media_info.py +0 -0
  67. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub.egg-info/dependency_links.txt +0 -0
  68. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub.egg-info/entry_points.txt +0 -0
  69. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub.egg-info/requires.txt +0 -0
  70. {parsehub-2.1.0 → parsehub-2.1.1}/src/parsehub.egg-info/top_level.txt +0 -0
  71. {parsehub-2.1.0 → parsehub-2.1.1}/test/test_cli.py +0 -0
  72. {parsehub-2.1.0 → parsehub-2.1.1}/test/test_cli_config.py +0 -0
  73. {parsehub-2.1.0 → parsehub-2.1.1}/test/test_downloader.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: parsehub
3
- Version: 2.1.0
3
+ Version: 2.1.1
4
4
  Summary: 轻量、异步、开箱即用的社交媒体聚合解析库
5
5
  Author-email: 梓澪 <zilingmio@gmail.com>
6
6
  License: MIT
@@ -70,26 +70,27 @@ Dynamic: license-file
70
70
 
71
71
  ## 🌐 支持平台
72
72
 
73
- | 平台 | 视频 | 图文 | 其他 |
74
- |-----------------|:--:|:--:|------|
75
- | **Twitter / X** | ✅ | ✅ | 📝 文章 |
76
- | **Instagram** | ✅ | ✅ | |
77
- | **YouTube** | ✅ | | 🎵 音乐 |
78
- | **Facebook** | ✅ | | |
79
- | **Threads** | ✅ | ✅ | |
80
- | **Bilibili** | ✅ | | 📝 动态 |
81
- | **抖音** | ✅ | ✅ | ☀️日常 |
82
- | **TikTok** | ✅ | ✅ | |
83
- | **微博** | ✅ | ✅ | |
84
- | **小红书** | ✅ | ✅ | |
85
- | **贴吧** | ✅ | ✅ | |
86
- | **微信公众号** | | ✅ | |
87
- | **快手** | ✅ | | |
88
- | **酷安** | | ✅ | |
89
- | **皮皮虾** | ✅ | ✅ | |
90
- | **最右** | ✅ | ✅ | |
91
- | **小黑盒** | ✅ | ✅ | |
92
- | **Snapchat** | ✅ | | |
73
+ | 平台 | 视频 | 图文 | 其他 |
74
+ |-----------------|:--:|:--:|---------------|
75
+ | **Twitter / X** | ✅ | ✅ | 📝 文章 |
76
+ | **Instagram** | ✅ | ✅ | |
77
+ | **YouTube** | ✅ | | 🎵 音乐 |
78
+ | **Facebook** | ✅ | | |
79
+ | **Threads** | ✅ | ✅ | |
80
+ | **Bilibili** | ✅ | | 📝 动态 |
81
+ | **抖音** | ✅ | ✅ | ☀️日常 |
82
+ | **TikTok** | ✅ | ✅ | |
83
+ | **微博** | ✅ | ✅ | |
84
+ | **小红书** | ✅ | ✅ | |
85
+ | **贴吧** | ✅ | ✅ | |
86
+ | **微信公众号** | | ✅ | |
87
+ | **快手** | ✅ | | |
88
+ | **酷安** | | ✅ | |
89
+ | **皮皮虾** | ✅ | ✅ | |
90
+ | **最右** | ✅ | ✅ | |
91
+ | **小黑盒** | ✅ | ✅ | |
92
+ | **Snapchat** | ✅ | | |
93
+ | **知乎** | ✅ | ✅ | 🐶 问答, 专栏, 圈子 |
93
94
 
94
95
  ## 📦 安装
95
96
 
@@ -210,6 +211,7 @@ print(result)
210
211
  - `TikTok`
211
212
  - `快手`
212
213
  - `小红书`
214
+ - `知乎`
213
215
 
214
216
  ```python
215
217
  from parsehub import ParseHub
@@ -315,6 +317,7 @@ except ParseError as exc:
315
317
  - [instaloader/instaloader](https://github.com/instaloader/instaloader)
316
318
  - [SocialSisterYi/bilibili-API-collect](https://github.com/SocialSisterYi/bilibili-API-collect)
317
319
  - [Nemo2011/bilibili-api](https://github.com/Nemo2011/bilibili-api)
320
+ - [cv-cat/ZhihuApis](https://github.com/cv-cat/ZhihuApis)
318
321
 
319
322
  ## 📜 开源协议
320
323
 
@@ -28,26 +28,27 @@
28
28
 
29
29
  ## 🌐 支持平台
30
30
 
31
- | 平台 | 视频 | 图文 | 其他 |
32
- |-----------------|:--:|:--:|------|
33
- | **Twitter / X** | ✅ | ✅ | 📝 文章 |
34
- | **Instagram** | ✅ | ✅ | |
35
- | **YouTube** | ✅ | | 🎵 音乐 |
36
- | **Facebook** | ✅ | | |
37
- | **Threads** | ✅ | ✅ | |
38
- | **Bilibili** | ✅ | | 📝 动态 |
39
- | **抖音** | ✅ | ✅ | ☀️日常 |
40
- | **TikTok** | ✅ | ✅ | |
41
- | **微博** | ✅ | ✅ | |
42
- | **小红书** | ✅ | ✅ | |
43
- | **贴吧** | ✅ | ✅ | |
44
- | **微信公众号** | | ✅ | |
45
- | **快手** | ✅ | | |
46
- | **酷安** | | ✅ | |
47
- | **皮皮虾** | ✅ | ✅ | |
48
- | **最右** | ✅ | ✅ | |
49
- | **小黑盒** | ✅ | ✅ | |
50
- | **Snapchat** | ✅ | | |
31
+ | 平台 | 视频 | 图文 | 其他 |
32
+ |-----------------|:--:|:--:|---------------|
33
+ | **Twitter / X** | ✅ | ✅ | 📝 文章 |
34
+ | **Instagram** | ✅ | ✅ | |
35
+ | **YouTube** | ✅ | | 🎵 音乐 |
36
+ | **Facebook** | ✅ | | |
37
+ | **Threads** | ✅ | ✅ | |
38
+ | **Bilibili** | ✅ | | 📝 动态 |
39
+ | **抖音** | ✅ | ✅ | ☀️日常 |
40
+ | **TikTok** | ✅ | ✅ | |
41
+ | **微博** | ✅ | ✅ | |
42
+ | **小红书** | ✅ | ✅ | |
43
+ | **贴吧** | ✅ | ✅ | |
44
+ | **微信公众号** | | ✅ | |
45
+ | **快手** | ✅ | | |
46
+ | **酷安** | | ✅ | |
47
+ | **皮皮虾** | ✅ | ✅ | |
48
+ | **最右** | ✅ | ✅ | |
49
+ | **小黑盒** | ✅ | ✅ | |
50
+ | **Snapchat** | ✅ | | |
51
+ | **知乎** | ✅ | ✅ | 🐶 问答, 专栏, 圈子 |
51
52
 
52
53
  ## 📦 安装
53
54
 
@@ -168,6 +169,7 @@ print(result)
168
169
  - `TikTok`
169
170
  - `快手`
170
171
  - `小红书`
172
+ - `知乎`
171
173
 
172
174
  ```python
173
175
  from parsehub import ParseHub
@@ -273,6 +275,7 @@ except ParseError as exc:
273
275
  - [instaloader/instaloader](https://github.com/instaloader/instaloader)
274
276
  - [SocialSisterYi/bilibili-API-collect](https://github.com/SocialSisterYi/bilibili-API-collect)
275
277
  - [Nemo2011/bilibili-api](https://github.com/Nemo2011/bilibili-api)
278
+ - [cv-cat/ZhihuApis](https://github.com/cv-cat/ZhihuApis)
276
279
 
277
280
  ## 📜 开源协议
278
281
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "parsehub"
3
- version = "2.1.0"
3
+ version = "2.1.1"
4
4
  description = "轻量、异步、开箱即用的社交媒体聚合解析库"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.12.0"
@@ -0,0 +1,64 @@
1
+ from ...provider_api.zhihu import ZhihuAPI, ZhihuPin, ZhihuPinType, ZhihuQA, ZhihuZhuanLan
2
+ from ...types import (
3
+ ImageParseResult,
4
+ ImageRef,
5
+ MultimediaParseResult,
6
+ Platform,
7
+ RichTextParseResult,
8
+ VideoParseResult,
9
+ VideoRef,
10
+ )
11
+ from ..base.base import BaseParser
12
+
13
+
14
+ class ZhihuParser(BaseParser):
15
+ __platform__ = Platform.ZHIHU
16
+ __supported_type__ = ["问答", "专栏", "圈子"]
17
+ __match__ = r"^(http(s)?://)?(www|zhuanlan).zhihu.com/(pin|question|p)/.*"
18
+
19
+ async def _do_parse(
20
+ self, raw_url: str
21
+ ) -> RichTextParseResult | MultimediaParseResult | ImageParseResult | VideoParseResult:
22
+ if not (c := self.cookie.get_value()):
23
+ raise ValueError("知乎需要配置已登录的 Cookie")
24
+ result = await ZhihuAPI(cookie=c, proxy=self.proxy).parse(raw_url)
25
+ match result:
26
+ case ZhihuQA():
27
+ if not result.markdown_answer:
28
+ return MultimediaParseResult(title=result.question)
29
+ return RichTextParseResult(
30
+ title=result.question,
31
+ media=[ImageRef(url=i) for i in result.imgs],
32
+ markdown_content=result.markdown_answer,
33
+ )
34
+ case ZhihuZhuanLan():
35
+ return RichTextParseResult(
36
+ title=result.title,
37
+ markdown_content=result.markdown_content,
38
+ media=[ImageRef(url=i) for i in result.imgs],
39
+ )
40
+ case ZhihuPin():
41
+ match result.type:
42
+ case ZhihuPinType.TEXT:
43
+ return ImageParseResult(title=result.title, content=result.plaintext_content)
44
+ case ZhihuPinType.IMAGE:
45
+ return ImageParseResult(
46
+ title=result.title,
47
+ content=result.plaintext_content,
48
+ photo=[
49
+ ImageRef(url=i.url, thumb_url=i.thumb_url, width=i.width, height=i.height)
50
+ for i in result.media
51
+ ],
52
+ )
53
+ case ZhihuPinType.VIDEO:
54
+ v = result.media[0]
55
+ return VideoParseResult(
56
+ title=result.title,
57
+ content=result.plaintext_content,
58
+ video=VideoRef(
59
+ url=v.url, thumb_url=v.thumb_url, height=v.height, width=v.width, duration=v.duration
60
+ ),
61
+ )
62
+
63
+
64
+ __all__ = ["ZhihuParser"]
@@ -0,0 +1,748 @@
1
+ # mypy: disable-error-code=no-untyped-def
2
+ """https://github.com/cv-cat/ZhihuApis
3
+ Pure-Python implementation of Zhihu's x-zse-96 (v2.0) signature algorithm.
4
+
5
+ This module reproduces, without Node.js / execjs, the encryption performed by
6
+ ``static/zhihu.js`` + ``static/other.js``.
7
+
8
+ Pipeline (reverse engineered from the obfuscated bytecode VM in ``other.js``):
9
+
10
+ 1. ``source = "+".join([zse93, path, d_c0, (body), (x_zst_81)])`` (empty parts skipped)
11
+ 2. ``digest = md5(source)`` (32 hex chars)
12
+ 3. ``plain = ascii(digest[14:])`` padded with PKCS#7 to 32 bytes
13
+ 4. ``cipher = SM4_CBC(plain, key=<fixed round keys>, iv=<random 16 bytes>)``
14
+ (the SM4 key schedule is baked in as ``SM4_ZK``; ``SM4_ZB`` is the S-box)
15
+ 5. ``blob = iv + cipher`` (48 bytes; the IV is prepended so the server can decrypt)
16
+ 6. ``sig = custom_base64(blob)`` where the 384-bit stream is split into 64
17
+ sextets, the sextet order is reversed and each sextet is XOR-ed with a fixed
18
+ periodic mask, then mapped through a permuted alphabet.
19
+ 7. ``x-zse-96 = "2.0_" + sig``
20
+
21
+ All constants below were extracted from the running JS and the mapping was
22
+ verified bit-exactly against 500 reference outputs.
23
+ """
24
+
25
+ import asyncio
26
+ import hashlib
27
+ import os
28
+ import re
29
+ from dataclasses import dataclass
30
+ from enum import Enum
31
+ from typing import Any, Self, cast
32
+ from urllib.parse import urlparse
33
+
34
+ import httpx
35
+
36
+ __all__ = ["get_x_zse_96", "ZhihuAPI", "ZhihuQA", "ZhihuZhuanLan", "ZhihuPin", "ZhihuPinType", "ZhihuMedia"]
37
+
38
+ from bs4 import BeautifulSoup
39
+ from markdown import markdown
40
+ from markdownify import MarkdownConverter
41
+
42
+
43
+ class ZhihuConverter(MarkdownConverter):
44
+ def convert_img(self, el: Any, text: Any, parent_tags: Any) -> str:
45
+ alt = el.attrs.get("alt", None) or ""
46
+ src = el.attrs.get("src", None) or ""
47
+ if src.startswith("data:image/svg"):
48
+ return alt
49
+ # if '/equation' in src:
50
+ # parsed_url = urlparse(src)
51
+ # query_params = parse_qs(parsed_url.query)
52
+ # tex_value = query_params.get('tex', [''])[0]
53
+ # src = f"https://latex.codecogs.com/png.image?\\dpi{{200}}\\bg{{white}}{tex_value}"
54
+ title = el.attrs.get("title", None) or ""
55
+ title_part = ' "{}"'.format(title.replace('"', r"\"")) if title else ""
56
+ options = cast(dict[str, Any], getattr(self, "options")) # noqa: B009
57
+ if "_inline" in parent_tags and el.parent.name not in options["keep_inline_images_in"]:
58
+ return alt
59
+
60
+ return f"![{alt}]({src}{title_part})"
61
+
62
+
63
+ def _zhihu_contenc_fmt(content: str) -> tuple[str, str, list[str]]:
64
+ soup = BeautifulSoup(content, "lxml")
65
+ markdown_content = ZhihuConverter(heading_style="ATX").convert(str(soup))
66
+ plaintext_content = "".join(BeautifulSoup(markdown(markdown_content), "lxml").find_all(string=True))
67
+ imgs = [
68
+ str(i["src"])
69
+ for i in soup.find_all("img")
70
+ if not str(i["src"]).startswith("data:image/svg") and "/equation" not in str(i["src"]) # 过滤掉 svg 和 equation
71
+ ]
72
+ return markdown_content, plaintext_content, imgs
73
+
74
+
75
+ @dataclass(kw_only=True)
76
+ class ZhihuQA:
77
+ question: str
78
+ imgs: list[str]
79
+ markdown_answer: str | None = None
80
+ plaintext_answer: str | None = None
81
+
82
+ @classmethod
83
+ def parse(cls, data: dict) -> Self:
84
+ question = data["question"]["title"]
85
+ answer = data["content"]
86
+ markdown_content, plaintext_content, imgs = _zhihu_contenc_fmt(answer)
87
+ return cls(question=question, imgs=imgs, markdown_answer=markdown_content, plaintext_answer=plaintext_content)
88
+
89
+
90
+ @dataclass(kw_only=True)
91
+ class ZhihuZhuanLan:
92
+ title: str
93
+ imgs: list[str]
94
+ markdown_content: str | None = None
95
+ plaintext_content: str | None = None
96
+
97
+ @classmethod
98
+ def parse(cls, data: dict) -> Self:
99
+ title = data["title"]
100
+ content = data["content"]
101
+ markdown_content, plaintext_content, imgs = _zhihu_contenc_fmt(content)
102
+ return cls(title=title, imgs=imgs, markdown_content=markdown_content, plaintext_content=plaintext_content)
103
+
104
+
105
+ class ZhihuPinType(Enum):
106
+ IMAGE = "image"
107
+ VIDEO = "video"
108
+ TEXT = "text"
109
+
110
+
111
+ @dataclass(kw_only=True)
112
+ class ZhihuMedia:
113
+ url: str
114
+ thumb_url: str | None = None
115
+ width: int = 0
116
+ height: int = 0
117
+ duration: int = 0
118
+
119
+
120
+ @dataclass(kw_only=True)
121
+ class ZhihuPin:
122
+ type: ZhihuPinType
123
+ title: str
124
+ media: list[ZhihuMedia]
125
+ markdown_content: str | None = None
126
+ plaintext_content: str | None = None
127
+
128
+ @classmethod
129
+ def parse(cls, result: dict) -> "ZhihuPin":
130
+ content: list = result["content"]
131
+ title = ""
132
+ text = ""
133
+ media = []
134
+ t = ZhihuPinType.TEXT
135
+ for c in content:
136
+ match c["type"]:
137
+ case "text":
138
+ title = c["title"]
139
+ text = c["content"]
140
+ case "image":
141
+ t = ZhihuPinType.IMAGE
142
+ media.append(
143
+ ZhihuMedia(
144
+ url=c["original_url"],
145
+ thumb_url=c["url"],
146
+ width=c["width"],
147
+ height=c["height"],
148
+ )
149
+ )
150
+ case "video":
151
+ t = ZhihuPinType.VIDEO
152
+ video_info = c["video_info"]
153
+ playlist: dict = video_info["playlist"]
154
+ v = list(playlist.values())[0] # 画质从高到低, 0为最高
155
+ media.append(
156
+ ZhihuMedia(
157
+ url=v["url"],
158
+ thumb_url=video_info["thumbnail"],
159
+ width=v["width"],
160
+ height=v["height"],
161
+ duration=video_info["duration"],
162
+ )
163
+ )
164
+ markdown_content, plaintext_content, imgs = _zhihu_contenc_fmt(text)
165
+ return cls(
166
+ title=title, media=media, markdown_content=markdown_content, plaintext_content=plaintext_content, type=t
167
+ )
168
+
169
+
170
+ class ZhihuAPI:
171
+ def __init__(self, cookie: dict[str, str], proxy: str | None = None):
172
+ self.proxy = proxy
173
+ self.cookie = cookie
174
+
175
+ @property
176
+ def d_c0(self) -> str:
177
+ v = self.cookie.get("d_c0")
178
+ if not v:
179
+ raise ValueError("d_c0 is not found in cookie")
180
+ return v
181
+
182
+ @staticmethod
183
+ def get_headers(x_zse_96: str) -> dict:
184
+ return {
185
+ "accept": "*/*",
186
+ "accept-language": "zh-CN,zh;q=0.9,en;q=0.8,en-GB;q=0.7,en-US;q=0.6",
187
+ "origin": "https://zhuanlan.zhihu.com",
188
+ "referer": "https://zhuanlan.zhihu.com/",
189
+ "sec-ch-ua": '"Microsoft Edge";v="123", "Not:A-Brand";v="8", "Chromium";v="123"',
190
+ "sec-ch-ua-mobile": "?0",
191
+ "sec-ch-ua-platform": '"Windows"',
192
+ "sec-fetch-dest": "empty",
193
+ "sec-fetch-mode": "cors",
194
+ "sec-fetch-site": "same-site",
195
+ "user-agent": (
196
+ "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko)"
197
+ " Chrome/123.0.0.0 Safari/537.36 Edg/123.0.0.0"
198
+ ),
199
+ "x-requested-with": "fetch",
200
+ "x-zse-93": "101_3_3.0",
201
+ "x-zse-96": x_zse_96,
202
+ }
203
+
204
+ async def parse(self, raw_url: str) -> ZhihuQA | ZhihuZhuanLan | ZhihuPin:
205
+ if "/question" in raw_url:
206
+ return await self.parse_qa(raw_url)
207
+ if "zhuanlan." in raw_url:
208
+ return await self.parse_zl(raw_url)
209
+ if "/pin/" in raw_url:
210
+ return await self.parse_pin(raw_url)
211
+ raise ValueError("不支持的类型")
212
+
213
+ async def parse_qa(self, raw_url: str) -> ZhihuQA:
214
+ qid, aid = self._get_qa_id(raw_url)
215
+ if aid:
216
+ result = await self._answers(aid)
217
+ return ZhihuQA.parse(result)
218
+ result = await self._questions_answers(qid)
219
+ data = result["data"]
220
+ if data:
221
+ result = await self._answers(data[0]["id"])
222
+ return ZhihuQA.parse(result)
223
+ result = await self._questions(qid)
224
+ return ZhihuQA(question=result["title"], imgs=[])
225
+
226
+ async def parse_zl(self, raw_url: str) -> ZhihuZhuanLan:
227
+ zl_id = self._get_zl_id(raw_url)
228
+ result = await self._zl(zl_id)
229
+ return ZhihuZhuanLan.parse(result)
230
+
231
+ async def parse_pin(self, raw_url: str) -> ZhihuPin:
232
+ pin_id = self._get_pin_id(raw_url)
233
+ result = await self._pin(pin_id)
234
+ return ZhihuPin.parse(result)
235
+
236
+ async def _questions(self, question_id: int | str) -> dict:
237
+ """获取问题"""
238
+ url = f"https://www.zhihu.com/api/v4/questions/{question_id}"
239
+ query: dict = {}
240
+ x_zse_96 = get_x_zse_96(url, query, self.d_c0)
241
+ headers = self.get_headers(x_zse_96)
242
+
243
+ async with httpx.AsyncClient() as client:
244
+ r = await client.get(url, headers=headers, params=query, cookies=self.cookie)
245
+ return dict(r.json())
246
+
247
+ async def _questions_answers(self, question_id: int | str) -> dict:
248
+ """获取问题的回答"""
249
+ url = f"https://www.zhihu.com/api/v4/questions/{question_id}/answers"
250
+ query = {
251
+ "offset": "",
252
+ "limit": "1",
253
+ "sort_by": "default",
254
+ "include": "data[*].content",
255
+ }
256
+ x_zse_96 = get_x_zse_96(url, query, self.d_c0)
257
+ headers = self.get_headers(x_zse_96)
258
+
259
+ async with httpx.AsyncClient() as client:
260
+ r = await client.get(url, headers=headers, params=query, cookies=self.cookie)
261
+ return dict(r.json())
262
+
263
+ async def _answers(self, answers_id: int | str) -> dict:
264
+ """获取问题的指定回答"""
265
+ url = f"https://www.zhihu.com/api/v4/answers/{answers_id}"
266
+ query = {
267
+ "include": "data[*].content",
268
+ }
269
+ x_zse_96 = get_x_zse_96(url, query, self.d_c0)
270
+ headers = self.get_headers(x_zse_96)
271
+
272
+ async with httpx.AsyncClient() as client:
273
+ r = await client.get(url, headers=headers, params=query, cookies=self.cookie)
274
+ return dict(r.json())
275
+
276
+ async def _zl(self, zl_id: int | str) -> dict:
277
+ url = f"https://zhuanlan.zhihu.com/api/articles/{zl_id}"
278
+ query: dict = {}
279
+ x_zse_96 = get_x_zse_96(url, query, self.d_c0)
280
+ headers = self.get_headers(x_zse_96)
281
+
282
+ async with httpx.AsyncClient() as client:
283
+ r = await client.get(url, headers=headers, params=query, cookies=self.cookie)
284
+ return dict(r.json())
285
+
286
+ async def _pin(self, pin_id: int | str) -> dict:
287
+ url = f"https://www.zhihu.com/api/v4/pins/{pin_id}"
288
+ query: dict = {}
289
+ x_zse_96 = get_x_zse_96(url, query, self.d_c0)
290
+ headers = self.get_headers(x_zse_96)
291
+
292
+ async with httpx.AsyncClient() as client:
293
+ r = await client.get(url, headers=headers, params=query, cookies=self.cookie)
294
+ return dict(r.json())
295
+
296
+ @staticmethod
297
+ def _get_qa_id(raw_url: str) -> tuple[str, str | None]:
298
+ """返回问题和回答 id, 没有回答 id 时返回 None"""
299
+ r = re.search(r"/question/(\d+)(?:/answer/(\d+))?", raw_url)
300
+ if not r:
301
+ raise ValueError("从链接中提取 问答 id 错误")
302
+ return r.group(1), r.group(2) if r.group(2) else None
303
+
304
+ @staticmethod
305
+ def _get_zl_id(raw_url: str) -> str:
306
+ r = re.search(r"/p/(\d+)", raw_url)
307
+ if not r:
308
+ raise ValueError("从链接中提取 专栏 id 错误")
309
+ return r.group(1)
310
+
311
+ @staticmethod
312
+ def _get_pin_id(raw_url: str) -> str:
313
+ r = re.search(r"/pin/(\d+)", raw_url)
314
+ if not r:
315
+ raise ValueError("从链接中提取 圈子 id 错误")
316
+ return r.group(1)
317
+
318
+
319
+ # --- SM4 constants (extracted from other.js VM) ---------------------------------
320
+ SM4_ZB = [
321
+ 20,
322
+ 223,
323
+ 245,
324
+ 7,
325
+ 248,
326
+ 2,
327
+ 194,
328
+ 209,
329
+ 87,
330
+ 6,
331
+ 227,
332
+ 253,
333
+ 240,
334
+ 128,
335
+ 222,
336
+ 91,
337
+ 237,
338
+ 9,
339
+ 125,
340
+ 157,
341
+ 230,
342
+ 93,
343
+ 252,
344
+ 205,
345
+ 90,
346
+ 79,
347
+ 144,
348
+ 199,
349
+ 159,
350
+ 197,
351
+ 186,
352
+ 167,
353
+ 39,
354
+ 37,
355
+ 156,
356
+ 198,
357
+ 38,
358
+ 42,
359
+ 43,
360
+ 168,
361
+ 217,
362
+ 153,
363
+ 15,
364
+ 103,
365
+ 80,
366
+ 189,
367
+ 71,
368
+ 191,
369
+ 97,
370
+ 84,
371
+ 247,
372
+ 95,
373
+ 36,
374
+ 69,
375
+ 14,
376
+ 35,
377
+ 12,
378
+ 171,
379
+ 28,
380
+ 114,
381
+ 178,
382
+ 148,
383
+ 86,
384
+ 182,
385
+ 32,
386
+ 83,
387
+ 158,
388
+ 109,
389
+ 22,
390
+ 255,
391
+ 94,
392
+ 238,
393
+ 151,
394
+ 85,
395
+ 77,
396
+ 124,
397
+ 254,
398
+ 18,
399
+ 4,
400
+ 26,
401
+ 123,
402
+ 176,
403
+ 232,
404
+ 193,
405
+ 131,
406
+ 172,
407
+ 143,
408
+ 142,
409
+ 150,
410
+ 30,
411
+ 10,
412
+ 146,
413
+ 162,
414
+ 62,
415
+ 224,
416
+ 218,
417
+ 196,
418
+ 229,
419
+ 1,
420
+ 192,
421
+ 213,
422
+ 27,
423
+ 110,
424
+ 56,
425
+ 231,
426
+ 180,
427
+ 138,
428
+ 107,
429
+ 242,
430
+ 187,
431
+ 54,
432
+ 120,
433
+ 19,
434
+ 44,
435
+ 117,
436
+ 228,
437
+ 215,
438
+ 203,
439
+ 53,
440
+ 239,
441
+ 251,
442
+ 127,
443
+ 81,
444
+ 11,
445
+ 133,
446
+ 96,
447
+ 204,
448
+ 132,
449
+ 41,
450
+ 115,
451
+ 73,
452
+ 55,
453
+ 249,
454
+ 147,
455
+ 102,
456
+ 48,
457
+ 122,
458
+ 145,
459
+ 106,
460
+ 118,
461
+ 74,
462
+ 190,
463
+ 29,
464
+ 16,
465
+ 174,
466
+ 5,
467
+ 177,
468
+ 129,
469
+ 63,
470
+ 113,
471
+ 99,
472
+ 31,
473
+ 161,
474
+ 76,
475
+ 246,
476
+ 34,
477
+ 211,
478
+ 13,
479
+ 60,
480
+ 68,
481
+ 207,
482
+ 160,
483
+ 65,
484
+ 111,
485
+ 82,
486
+ 165,
487
+ 67,
488
+ 169,
489
+ 225,
490
+ 57,
491
+ 112,
492
+ 244,
493
+ 155,
494
+ 51,
495
+ 236,
496
+ 200,
497
+ 233,
498
+ 58,
499
+ 61,
500
+ 47,
501
+ 100,
502
+ 137,
503
+ 185,
504
+ 64,
505
+ 17,
506
+ 70,
507
+ 234,
508
+ 163,
509
+ 219,
510
+ 108,
511
+ 170,
512
+ 166,
513
+ 59,
514
+ 149,
515
+ 52,
516
+ 105,
517
+ 24,
518
+ 212,
519
+ 78,
520
+ 173,
521
+ 45,
522
+ 0,
523
+ 116,
524
+ 226,
525
+ 119,
526
+ 136,
527
+ 206,
528
+ 135,
529
+ 175,
530
+ 195,
531
+ 25,
532
+ 92,
533
+ 121,
534
+ 208,
535
+ 126,
536
+ 139,
537
+ 3,
538
+ 75,
539
+ 141,
540
+ 21,
541
+ 130,
542
+ 98,
543
+ 241,
544
+ 40,
545
+ 154,
546
+ 66,
547
+ 184,
548
+ 49,
549
+ 181,
550
+ 46,
551
+ 243,
552
+ 88,
553
+ 101,
554
+ 183,
555
+ 8,
556
+ 23,
557
+ 72,
558
+ 188,
559
+ 104,
560
+ 179,
561
+ 210,
562
+ 134,
563
+ 250,
564
+ 201,
565
+ 164,
566
+ 89,
567
+ 216,
568
+ 202,
569
+ 220,
570
+ 50,
571
+ 221,
572
+ 152,
573
+ 140,
574
+ 33,
575
+ 235,
576
+ 214,
577
+ ]
578
+ # Pre-expanded 32 SM4 round keys (already derived from the fixed encryption key).
579
+ SM4_ZK = [
580
+ k & 0xFFFFFFFF
581
+ for k in [
582
+ 1170614578,
583
+ 1024848638,
584
+ 1413669199,
585
+ -343334464,
586
+ -766094290,
587
+ -1373058082,
588
+ -143119608,
589
+ -297228157,
590
+ 1933479194,
591
+ -971186181,
592
+ -406453910,
593
+ 460404854,
594
+ -547427574,
595
+ -1891326262,
596
+ -1679095901,
597
+ 2119585428,
598
+ -2029270069,
599
+ 2035090028,
600
+ -1521520070,
601
+ -5587175,
602
+ -77751101,
603
+ -2094365853,
604
+ -1243052806,
605
+ 1579901135,
606
+ 1321810770,
607
+ 456816404,
608
+ -1391643889,
609
+ -229302305,
610
+ 330002838,
611
+ -788960546,
612
+ 363569021,
613
+ -1947871109,
614
+ ]
615
+ ]
616
+
617
+ # --- custom base64 constants (extracted / solved from other.js) -----------------
618
+ ALPHABET = "6fpLRqJO8M/c3jnYxFkUVC4ZIG12SiH=5v0mXDazWBTsuw7QetbKdoPyAl+hN9rgE"
619
+ # per-sextet XOR mask, period 16 groups, repeated 4 times over the 64 output chars
620
+ MASKS = [58, 0, 0, 0, 0, 40, 3, 0, 0, 0, 32, 14, 0, 0, 0, 0] * 4
621
+
622
+ ZSE93 = "101_3_3.0"
623
+
624
+
625
+ def _u32(x):
626
+ return x & 0xFFFFFFFF
627
+
628
+
629
+ def _rotl(x, n):
630
+ return _u32((x << n) | (x >> (32 - n)))
631
+
632
+
633
+ def _load_be(arr, o):
634
+ return _u32((arr[o] << 24) | (arr[o + 1] << 16) | (arr[o + 2] << 8) | arr[o + 3])
635
+
636
+
637
+ def _store_be(v, arr, o):
638
+ arr[o] = (v >> 24) & 255
639
+ arr[o + 1] = (v >> 16) & 255
640
+ arr[o + 2] = (v >> 8) & 255
641
+ arr[o + 3] = v & 255
642
+
643
+
644
+ def _tau_l(x):
645
+ b = [(x >> 24) & 255, (x >> 16) & 255, (x >> 8) & 255, x & 255]
646
+ t = [SM4_ZB[b[0]], SM4_ZB[b[1]], SM4_ZB[b[2]], SM4_ZB[b[3]]]
647
+ ec = _load_be(t, 0)
648
+ return _u32(ec ^ _rotl(ec, 2) ^ _rotl(ec, 10) ^ _rotl(ec, 18) ^ _rotl(ec, 24))
649
+
650
+
651
+ def _sm4_encrypt_block(block):
652
+ out = [0] * 16
653
+ x = [0] * 36
654
+ x[0] = _load_be(block, 0)
655
+ x[1] = _load_be(block, 4)
656
+ x[2] = _load_be(block, 8)
657
+ x[3] = _load_be(block, 12)
658
+ for i in range(32):
659
+ x[i + 4] = _u32(x[i] ^ _tau_l(x[i + 1] ^ x[i + 2] ^ x[i + 3] ^ SM4_ZK[i]))
660
+ _store_be(x[35], out, 0)
661
+ _store_be(x[34], out, 4)
662
+ _store_be(x[33], out, 8)
663
+ _store_be(x[32], out, 12)
664
+ return out
665
+
666
+
667
+ def _sm4_cbc(data, iv):
668
+ result = []
669
+ prev = list(iv)
670
+ for off in range(0, len(data), 16):
671
+ block = data[off : off + 16]
672
+ xored = [block[i] ^ prev[i] for i in range(16)]
673
+ prev = _sm4_encrypt_block(xored)
674
+ result.extend(prev)
675
+ return result
676
+
677
+
678
+ def _pkcs7(data, block=16):
679
+ pad = block - (len(data) % block)
680
+ return list(data) + [pad] * pad
681
+
682
+
683
+ def _custom_b64(blob):
684
+ # 48 bytes -> 384 bits -> 64 sextets
685
+ bits = []
686
+ for b in blob:
687
+ for j in range(8):
688
+ bits.append((b >> (7 - j)) & 1)
689
+ in_sext = []
690
+ for k in range(64):
691
+ v = 0
692
+ for j in range(6):
693
+ v = (v << 1) | bits[6 * k + j]
694
+ in_sext.append(v)
695
+ out = []
696
+ for g in range(64):
697
+ out.append(ALPHABET[in_sext[63 - g] ^ MASKS[g]])
698
+ return "".join(out)
699
+
700
+
701
+ def zhihu_encrypt(digest, iv=None):
702
+ """Encrypt a 32-char md5 hex string, returning the base64 signature body.
703
+
704
+ ``iv`` (16 bytes) is prepended to the ciphertext; the server recovers it, so
705
+ any value works. Defaults to random bytes to mimic the browser.
706
+ """
707
+ if iv is None:
708
+ iv = list(os.urandom(16))
709
+ else:
710
+ iv = list(iv)
711
+ plain = _pkcs7([ord(c) for c in digest[14:]])
712
+ cipher = _sm4_cbc(plain, iv)
713
+ return _custom_b64(iv + cipher)
714
+
715
+
716
+ def encrypt_md5(source, iv=None):
717
+ """md5(source) then encrypt -> signature body (without the ``2.0_`` prefix)."""
718
+ digest = hashlib.md5(source.encode("utf-8")).hexdigest()
719
+ return zhihu_encrypt(digest, iv=iv)
720
+
721
+
722
+ def get_x_zse_96(url, params, d_c0, body="", x_zst_81=None, iv=None):
723
+ """Compute the full ``x-zse-96`` header value (pure Python, no Node)."""
724
+ if params:
725
+ query = "&".join(f"{k}={v}" for k, v in params.items())
726
+ er = url + "?" + query
727
+ else:
728
+ er = url
729
+ parsed = urlparse(er)
730
+ path = parsed.path + (("?" + parsed.query) if parsed.query else "")
731
+ parts = [ZSE93, path, d_c0]
732
+ if body:
733
+ parts.append(body)
734
+ if x_zst_81:
735
+ parts.append(x_zst_81)
736
+ source = "+".join(parts)
737
+ return "2.0_" + encrypt_md5(source, iv=iv)
738
+
739
+
740
+ if __name__ == "__main__":
741
+
742
+ async def main() -> None:
743
+ cookies = ""
744
+ cookie = {i.split("=")[0]: "=".join(i.split("=")[1:]) for i in cookies.split("; ")}
745
+ zhihu = ZhihuAPI(cookie=cookie)
746
+ print(await zhihu.parse("https://www.zhihu.com/pin/2052700331017080947"))
747
+
748
+ asyncio.run(main())
@@ -22,6 +22,7 @@ class Platform(Enum):
22
22
  YOUTUBE = ("youtube", "Youtube")
23
23
  ZUIYOU = ("zuiyou", "最右")
24
24
  SNAPCHAT = ("snapchat", "Snapchat")
25
+ ZHIHU = ("zhihu", "知乎")
25
26
 
26
27
  def __init__(self, platform_id: str, platform_name: str) -> None:
27
28
  self.id = platform_id
@@ -101,7 +101,7 @@ class ParseResult(ABC): # noqa: B024
101
101
  :return: DownloadResult
102
102
  """
103
103
  if self.media is None:
104
- raise DownloadError("没有可下载的媒体")
104
+ return DownloadResult(output_dir=output_dir, media=[])
105
105
  media_list = list(self.media) if isinstance(self.media, Sequence) else [self.media]
106
106
  is_single = not isinstance(self.media, Sequence)
107
107
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: parsehub
3
- Version: 2.1.0
3
+ Version: 2.1.1
4
4
  Summary: 轻量、异步、开箱即用的社交媒体聚合解析库
5
5
  Author-email: 梓澪 <zilingmio@gmail.com>
6
6
  License: MIT
@@ -70,26 +70,27 @@ Dynamic: license-file
70
70
 
71
71
  ## 🌐 支持平台
72
72
 
73
- | 平台 | 视频 | 图文 | 其他 |
74
- |-----------------|:--:|:--:|------|
75
- | **Twitter / X** | ✅ | ✅ | 📝 文章 |
76
- | **Instagram** | ✅ | ✅ | |
77
- | **YouTube** | ✅ | | 🎵 音乐 |
78
- | **Facebook** | ✅ | | |
79
- | **Threads** | ✅ | ✅ | |
80
- | **Bilibili** | ✅ | | 📝 动态 |
81
- | **抖音** | ✅ | ✅ | ☀️日常 |
82
- | **TikTok** | ✅ | ✅ | |
83
- | **微博** | ✅ | ✅ | |
84
- | **小红书** | ✅ | ✅ | |
85
- | **贴吧** | ✅ | ✅ | |
86
- | **微信公众号** | | ✅ | |
87
- | **快手** | ✅ | | |
88
- | **酷安** | | ✅ | |
89
- | **皮皮虾** | ✅ | ✅ | |
90
- | **最右** | ✅ | ✅ | |
91
- | **小黑盒** | ✅ | ✅ | |
92
- | **Snapchat** | ✅ | | |
73
+ | 平台 | 视频 | 图文 | 其他 |
74
+ |-----------------|:--:|:--:|---------------|
75
+ | **Twitter / X** | ✅ | ✅ | 📝 文章 |
76
+ | **Instagram** | ✅ | ✅ | |
77
+ | **YouTube** | ✅ | | 🎵 音乐 |
78
+ | **Facebook** | ✅ | | |
79
+ | **Threads** | ✅ | ✅ | |
80
+ | **Bilibili** | ✅ | | 📝 动态 |
81
+ | **抖音** | ✅ | ✅ | ☀️日常 |
82
+ | **TikTok** | ✅ | ✅ | |
83
+ | **微博** | ✅ | ✅ | |
84
+ | **小红书** | ✅ | ✅ | |
85
+ | **贴吧** | ✅ | ✅ | |
86
+ | **微信公众号** | | ✅ | |
87
+ | **快手** | ✅ | | |
88
+ | **酷安** | | ✅ | |
89
+ | **皮皮虾** | ✅ | ✅ | |
90
+ | **最右** | ✅ | ✅ | |
91
+ | **小黑盒** | ✅ | ✅ | |
92
+ | **Snapchat** | ✅ | | |
93
+ | **知乎** | ✅ | ✅ | 🐶 问答, 专栏, 圈子 |
93
94
 
94
95
  ## 📦 安装
95
96
 
@@ -210,6 +211,7 @@ print(result)
210
211
  - `TikTok`
211
212
  - `快手`
212
213
  - `小红书`
214
+ - `知乎`
213
215
 
214
216
  ```python
215
217
  from parsehub import ParseHub
@@ -315,6 +317,7 @@ except ParseError as exc:
315
317
  - [instaloader/instaloader](https://github.com/instaloader/instaloader)
316
318
  - [SocialSisterYi/bilibili-API-collect](https://github.com/SocialSisterYi/bilibili-API-collect)
317
319
  - [Nemo2011/bilibili-api](https://github.com/Nemo2011/bilibili-api)
320
+ - [cv-cat/ZhihuApis](https://github.com/cv-cat/ZhihuApis)
318
321
 
319
322
  ## 📜 开源协议
320
323
 
@@ -36,6 +36,7 @@ src/parsehub/parsers/parser/weixin.py
36
36
  src/parsehub/parsers/parser/xhs.py
37
37
  src/parsehub/parsers/parser/xiaoheihe.py
38
38
  src/parsehub/parsers/parser/youtube.py
39
+ src/parsehub/parsers/parser/zhihu.py
39
40
  src/parsehub/parsers/parser/zuiyou.py
40
41
  src/parsehub/provider_api/__init__.py
41
42
  src/parsehub/provider_api/bilibili.py
@@ -52,6 +53,7 @@ src/parsehub/provider_api/weibo.py
52
53
  src/parsehub/provider_api/weixin.py
53
54
  src/parsehub/provider_api/xhs.py
54
55
  src/parsehub/provider_api/xiaoheihe.py
56
+ src/parsehub/provider_api/zhihu.py
55
57
  src/parsehub/provider_api/zuiyou.py
56
58
  src/parsehub/types/__init__.py
57
59
  src/parsehub/types/callback.py
@@ -402,6 +402,12 @@ class TestPlatformUrlMatching(unittest.TestCase):
402
402
  "https://www.snapchat.com/@snapchat/spotlight/W7_EDlXWTBiXAEEniNoMPwAAYbHBpemNsYmlyAZ7mTxgqAZ7mTuxMAAAAAw",
403
403
  "https://www.snapchat.com/@creativemindsho/gBRYnSexSxSBqXdq2Y6bhAAAga2djanpnd3JlAZ8fYED8AZ8fYD7pAAAAAA",
404
404
  ],
405
+ Platform.ZHIHU: [
406
+ "https://www.zhihu.com/pin/2050216877939482871",
407
+ "https://www.zhihu.com/question/2057559076813510452",
408
+ "https://zhuanlan.zhihu.com/p/1989096494578558904",
409
+ "https://www.zhihu.com/question/597674895/answer/3004370705",
410
+ ],
405
411
  }
406
412
 
407
413
  for platform, urls in cases.items():
File without changes
File without changes
File without changes
File without changes
File without changes