wechat-article-parser 0.0.3__tar.gz → 0.0.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {wechat_article_parser-0.0.3 → wechat_article_parser-0.0.5}/PKG-INFO +48 -1
- {wechat_article_parser-0.0.3 → wechat_article_parser-0.0.5}/README.md +47 -0
- {wechat_article_parser-0.0.3 → wechat_article_parser-0.0.5}/pyproject.toml +2 -1
- {wechat_article_parser-0.0.3 → wechat_article_parser-0.0.5}/src/wechat_article_parser/models.py +3 -0
- {wechat_article_parser-0.0.3 → wechat_article_parser-0.0.5}/src/wechat_article_parser/parser.py +37 -6
- {wechat_article_parser-0.0.3 → wechat_article_parser-0.0.5}/tests/conftest.py +6 -0
- {wechat_article_parser-0.0.3 → wechat_article_parser-0.0.5}/tests/test_parser.py +24 -10
- {wechat_article_parser-0.0.3 → wechat_article_parser-0.0.5}/.gitignore +0 -0
- {wechat_article_parser-0.0.3 → wechat_article_parser-0.0.5}/LICENSE +0 -0
- {wechat_article_parser-0.0.3 → wechat_article_parser-0.0.5}/src/wechat_article_parser/__init__.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: wechat-article-parser
|
|
3
|
-
Version: 0.0.
|
|
3
|
+
Version: 0.0.5
|
|
4
4
|
Summary: WeChat MP article parser - extract metadata and content from WeChat public account articles
|
|
5
5
|
License-Expression: MIT
|
|
6
6
|
License-File: LICENSE
|
|
@@ -75,12 +75,16 @@ print(result.article_markdown)
|
|
|
75
75
|
|
|
76
76
|
- `timeout`:请求超时时间,单位秒,默认 15
|
|
77
77
|
- `user_agent`:自定义 User-Agent,不传则使用内置默认值
|
|
78
|
+
- `proxy`:HTTP/HTTPS 代理地址,不传则直连
|
|
79
|
+
- `include_raw_html`:是否在返回结果中带上原始 HTML 网页源代码,默认 `False`
|
|
78
80
|
|
|
79
81
|
```python
|
|
80
82
|
result = parse(
|
|
81
83
|
"https://mp.weixin.qq.com/s/xxxxx",
|
|
82
84
|
timeout=30,
|
|
83
85
|
user_agent="MyBot/1.0",
|
|
86
|
+
proxy="http://user:pass@127.0.0.1:7890",
|
|
87
|
+
include_raw_html=True,
|
|
84
88
|
)
|
|
85
89
|
```
|
|
86
90
|
|
|
@@ -96,6 +100,7 @@ result = parse(
|
|
|
96
100
|
| `mp_alias` | `str` | 公众号别名 |
|
|
97
101
|
| `mp_image` | `str` | 公众号头像链接 |
|
|
98
102
|
| `mp_description` | `str` | 公众号简介 |
|
|
103
|
+
| `mp_account_type` | `AccountType` | 账号类型:`AccountType.SUBSCRIPTION`(订阅号)/ `AccountType.SERVICE`(服务号)/ `AccountType.UNKNOWN`(未识别) |
|
|
99
104
|
| `article_id` | `str` | 文章 ID |
|
|
100
105
|
| `article_msg_id` | `int` | 文章所在的群发消息 ID |
|
|
101
106
|
| `article_idx` | `int` | 群发图文中的位置(从 1 开始) |
|
|
@@ -105,11 +110,33 @@ result = parse(
|
|
|
105
110
|
| `article_description` | `str` | 文章摘要 |
|
|
106
111
|
| `article_markdown` | `str` | 文章正文的 Markdown 内容 |
|
|
107
112
|
| `article_publish_time` | `int` | 发布时间(Unix 时间戳) |
|
|
113
|
+
| `raw_html` | `str` | 原始 HTML 网页源代码(仅在 `include_raw_html=True` 时填充,否则为空字符串) |
|
|
108
114
|
| `images` | `list[str]` | 文章中提取的所有图片链接 |
|
|
109
115
|
| `is_valid` | `bool` | 关键字段是否全部解析成功(属性) |
|
|
110
116
|
|
|
111
117
|
`is_valid` 为 `True` 的条件:`mp_id`、`mp_name`、`article_id`、`article_msg_id`、`article_idx`、`article_sn`、`article_title`、`article_markdown`、`article_publish_time` 均不为空/零。
|
|
112
118
|
|
|
119
|
+
### 判断账号类型
|
|
120
|
+
|
|
121
|
+
`AccountType` 继承自 `str` 枚举,既支持枚举比较,也支持与中文字符串直接比较:
|
|
122
|
+
|
|
123
|
+
```python
|
|
124
|
+
from wechat_article_parser import parse, AccountType
|
|
125
|
+
|
|
126
|
+
result = parse("https://mp.weixin.qq.com/s/xxxxx")
|
|
127
|
+
|
|
128
|
+
# 推荐:枚举比较(有类型提示与 IDE 补全)
|
|
129
|
+
if result.mp_account_type == AccountType.SERVICE:
|
|
130
|
+
print("这是服务号")
|
|
131
|
+
|
|
132
|
+
# 也支持:字符串字面量比较
|
|
133
|
+
if result.mp_account_type == "服务号":
|
|
134
|
+
print("这是服务号")
|
|
135
|
+
|
|
136
|
+
# 打印直接输出中文值
|
|
137
|
+
print(f"账号类型: {result.mp_account_type}") # 账号类型: 订阅号
|
|
138
|
+
```
|
|
139
|
+
|
|
113
140
|
## 异常处理
|
|
114
141
|
|
|
115
142
|
### WeChatVerifyError
|
|
@@ -181,3 +208,23 @@ pytest tests/test_parser.py::test_fetch_all -s --url "https://mp.weixin.qq.com/s
|
|
|
181
208
|
```bash
|
|
182
209
|
pytest tests/test_parser.py::test_fetch_markdown -s --url "https://mp.weixin.qq.com/s/xxxxx"
|
|
183
210
|
```
|
|
211
|
+
|
|
212
|
+
### 测试单个链接 - 打印原始 HTML 网页源代码
|
|
213
|
+
|
|
214
|
+
```bash
|
|
215
|
+
pytest tests/test_parser.py::test_fetch_raw_html -s --url "https://mp.weixin.qq.com/s/xxxxx"
|
|
216
|
+
```
|
|
217
|
+
|
|
218
|
+
### 通过 HTTP 代理运行测试
|
|
219
|
+
|
|
220
|
+
所有测试命令都支持 `--proxy` 参数,传入后所有请求都会经由该代理;不传则直连。常用于 IP 被微信限流时换出口:
|
|
221
|
+
|
|
222
|
+
```bash
|
|
223
|
+
# 全量测试走代理
|
|
224
|
+
pytest tests/test_parser.py -v -s --proxy "http://127.0.0.1:7890"
|
|
225
|
+
|
|
226
|
+
# 测试单个链接走代理
|
|
227
|
+
pytest tests/test_parser.py::test_fetch_all -s \
|
|
228
|
+
--url "https://mp.weixin.qq.com/s/xxxxx" \
|
|
229
|
+
--proxy "http://user:pass@127.0.0.1:7890"
|
|
230
|
+
```
|
|
@@ -60,12 +60,16 @@ print(result.article_markdown)
|
|
|
60
60
|
|
|
61
61
|
- `timeout`:请求超时时间,单位秒,默认 15
|
|
62
62
|
- `user_agent`:自定义 User-Agent,不传则使用内置默认值
|
|
63
|
+
- `proxy`:HTTP/HTTPS 代理地址,不传则直连
|
|
64
|
+
- `include_raw_html`:是否在返回结果中带上原始 HTML 网页源代码,默认 `False`
|
|
63
65
|
|
|
64
66
|
```python
|
|
65
67
|
result = parse(
|
|
66
68
|
"https://mp.weixin.qq.com/s/xxxxx",
|
|
67
69
|
timeout=30,
|
|
68
70
|
user_agent="MyBot/1.0",
|
|
71
|
+
proxy="http://user:pass@127.0.0.1:7890",
|
|
72
|
+
include_raw_html=True,
|
|
69
73
|
)
|
|
70
74
|
```
|
|
71
75
|
|
|
@@ -81,6 +85,7 @@ result = parse(
|
|
|
81
85
|
| `mp_alias` | `str` | 公众号别名 |
|
|
82
86
|
| `mp_image` | `str` | 公众号头像链接 |
|
|
83
87
|
| `mp_description` | `str` | 公众号简介 |
|
|
88
|
+
| `mp_account_type` | `AccountType` | 账号类型:`AccountType.SUBSCRIPTION`(订阅号)/ `AccountType.SERVICE`(服务号)/ `AccountType.UNKNOWN`(未识别) |
|
|
84
89
|
| `article_id` | `str` | 文章 ID |
|
|
85
90
|
| `article_msg_id` | `int` | 文章所在的群发消息 ID |
|
|
86
91
|
| `article_idx` | `int` | 群发图文中的位置(从 1 开始) |
|
|
@@ -90,11 +95,33 @@ result = parse(
|
|
|
90
95
|
| `article_description` | `str` | 文章摘要 |
|
|
91
96
|
| `article_markdown` | `str` | 文章正文的 Markdown 内容 |
|
|
92
97
|
| `article_publish_time` | `int` | 发布时间(Unix 时间戳) |
|
|
98
|
+
| `raw_html` | `str` | 原始 HTML 网页源代码(仅在 `include_raw_html=True` 时填充,否则为空字符串) |
|
|
93
99
|
| `images` | `list[str]` | 文章中提取的所有图片链接 |
|
|
94
100
|
| `is_valid` | `bool` | 关键字段是否全部解析成功(属性) |
|
|
95
101
|
|
|
96
102
|
`is_valid` 为 `True` 的条件:`mp_id`、`mp_name`、`article_id`、`article_msg_id`、`article_idx`、`article_sn`、`article_title`、`article_markdown`、`article_publish_time` 均不为空/零。
|
|
97
103
|
|
|
104
|
+
### 判断账号类型
|
|
105
|
+
|
|
106
|
+
`AccountType` 继承自 `str` 枚举,既支持枚举比较,也支持与中文字符串直接比较:
|
|
107
|
+
|
|
108
|
+
```python
|
|
109
|
+
from wechat_article_parser import parse, AccountType
|
|
110
|
+
|
|
111
|
+
result = parse("https://mp.weixin.qq.com/s/xxxxx")
|
|
112
|
+
|
|
113
|
+
# 推荐:枚举比较(有类型提示与 IDE 补全)
|
|
114
|
+
if result.mp_account_type == AccountType.SERVICE:
|
|
115
|
+
print("这是服务号")
|
|
116
|
+
|
|
117
|
+
# 也支持:字符串字面量比较
|
|
118
|
+
if result.mp_account_type == "服务号":
|
|
119
|
+
print("这是服务号")
|
|
120
|
+
|
|
121
|
+
# 打印直接输出中文值
|
|
122
|
+
print(f"账号类型: {result.mp_account_type}") # 账号类型: 订阅号
|
|
123
|
+
```
|
|
124
|
+
|
|
98
125
|
## 异常处理
|
|
99
126
|
|
|
100
127
|
### WeChatVerifyError
|
|
@@ -166,3 +193,23 @@ pytest tests/test_parser.py::test_fetch_all -s --url "https://mp.weixin.qq.com/s
|
|
|
166
193
|
```bash
|
|
167
194
|
pytest tests/test_parser.py::test_fetch_markdown -s --url "https://mp.weixin.qq.com/s/xxxxx"
|
|
168
195
|
```
|
|
196
|
+
|
|
197
|
+
### 测试单个链接 - 打印原始 HTML 网页源代码
|
|
198
|
+
|
|
199
|
+
```bash
|
|
200
|
+
pytest tests/test_parser.py::test_fetch_raw_html -s --url "https://mp.weixin.qq.com/s/xxxxx"
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
### 通过 HTTP 代理运行测试
|
|
204
|
+
|
|
205
|
+
所有测试命令都支持 `--proxy` 参数,传入后所有请求都会经由该代理;不传则直连。常用于 IP 被微信限流时换出口:
|
|
206
|
+
|
|
207
|
+
```bash
|
|
208
|
+
# 全量测试走代理
|
|
209
|
+
pytest tests/test_parser.py -v -s --proxy "http://127.0.0.1:7890"
|
|
210
|
+
|
|
211
|
+
# 测试单个链接走代理
|
|
212
|
+
pytest tests/test_parser.py::test_fetch_all -s \
|
|
213
|
+
--url "https://mp.weixin.qq.com/s/xxxxx" \
|
|
214
|
+
--proxy "http://user:pass@127.0.0.1:7890"
|
|
215
|
+
```
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "wechat-article-parser"
|
|
7
|
-
version = "0.0.
|
|
7
|
+
version = "0.0.5"
|
|
8
8
|
description = "WeChat MP article parser - extract metadata and content from WeChat public account articles"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -26,3 +26,4 @@ packages = ["src/wechat_article_parser"]
|
|
|
26
26
|
|
|
27
27
|
[tool.pytest.ini_options]
|
|
28
28
|
asyncio_mode = "auto"
|
|
29
|
+
pythonpath = ["src"]
|
{wechat_article_parser-0.0.3 → wechat_article_parser-0.0.5}/src/wechat_article_parser/parser.py
RENAMED
|
@@ -476,30 +476,57 @@ def _parse_html(url: str, html: str) -> ArticleResult:
|
|
|
476
476
|
# 公开接口
|
|
477
477
|
# ---------------------------------------------------------------------------
|
|
478
478
|
|
|
479
|
-
def parse(
|
|
479
|
+
def parse(
|
|
480
|
+
url: str,
|
|
481
|
+
*,
|
|
482
|
+
timeout: int = _TIMEOUT,
|
|
483
|
+
user_agent: str | None = None,
|
|
484
|
+
proxy: str | None = None,
|
|
485
|
+
include_raw_html: bool = False,
|
|
486
|
+
) -> ArticleResult:
|
|
480
487
|
"""抓取并解析微信公众号文章(同步方式)。
|
|
481
488
|
|
|
482
489
|
Args:
|
|
483
490
|
url: 微信文章链接。
|
|
484
491
|
timeout: 请求超时时间(秒)。
|
|
485
492
|
user_agent: 自定义 User-Agent,不传则使用内置默认值。
|
|
493
|
+
proxy: HTTP/HTTPS 代理地址,例如 "http://user:pass@host:port",不传则直连。
|
|
494
|
+
include_raw_html: 是否在结果中返回原始 HTML 网页源代码,默认不返回。
|
|
486
495
|
|
|
487
496
|
Returns:
|
|
488
497
|
包含解析数据的 ArticleResult。
|
|
489
498
|
"""
|
|
490
499
|
ua = user_agent or _USER_AGENT
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
500
|
+
with httpx.Client(
|
|
501
|
+
headers={"User-Agent": ua},
|
|
502
|
+
timeout=timeout,
|
|
503
|
+
follow_redirects=True,
|
|
504
|
+
proxy=proxy,
|
|
505
|
+
) as client:
|
|
506
|
+
response = client.get(url)
|
|
507
|
+
response.raise_for_status()
|
|
508
|
+
result = _parse_html(url, response.text)
|
|
509
|
+
if include_raw_html:
|
|
510
|
+
result.raw_html = response.text
|
|
511
|
+
return result
|
|
494
512
|
|
|
495
513
|
|
|
496
|
-
async def parse_async(
|
|
514
|
+
async def parse_async(
|
|
515
|
+
url: str,
|
|
516
|
+
*,
|
|
517
|
+
timeout: int = _TIMEOUT,
|
|
518
|
+
user_agent: str | None = None,
|
|
519
|
+
proxy: str | None = None,
|
|
520
|
+
include_raw_html: bool = False,
|
|
521
|
+
) -> ArticleResult:
|
|
497
522
|
"""抓取并解析微信公众号文章(异步方式)。
|
|
498
523
|
|
|
499
524
|
Args:
|
|
500
525
|
url: 微信文章链接。
|
|
501
526
|
timeout: 请求超时时间(秒)。
|
|
502
527
|
user_agent: 自定义 User-Agent,不传则使用内置默认值。
|
|
528
|
+
proxy: HTTP/HTTPS 代理地址,例如 "http://user:pass@host:port",不传则直连。
|
|
529
|
+
include_raw_html: 是否在结果中返回原始 HTML 网页源代码,默认不返回。
|
|
503
530
|
|
|
504
531
|
Returns:
|
|
505
532
|
包含解析数据的 ArticleResult。
|
|
@@ -509,7 +536,11 @@ async def parse_async(url: str, *, timeout: int = _TIMEOUT, user_agent: str | No
|
|
|
509
536
|
headers={"User-Agent": ua},
|
|
510
537
|
timeout=timeout,
|
|
511
538
|
follow_redirects=True,
|
|
539
|
+
proxy=proxy,
|
|
512
540
|
) as client:
|
|
513
541
|
response = await client.get(url)
|
|
514
542
|
response.raise_for_status()
|
|
515
|
-
|
|
543
|
+
result = _parse_html(url, response.text)
|
|
544
|
+
if include_raw_html:
|
|
545
|
+
result.raw_html = response.text
|
|
546
|
+
return result
|
|
@@ -3,6 +3,7 @@ import pytest
|
|
|
3
3
|
|
|
4
4
|
def pytest_addoption(parser):
|
|
5
5
|
parser.addoption("--url", default=None, help="微信公众号文章链接")
|
|
6
|
+
parser.addoption("--proxy", default=None, help="HTTP 代理地址,例如 http://127.0.0.1:7890")
|
|
6
7
|
|
|
7
8
|
|
|
8
9
|
@pytest.fixture
|
|
@@ -11,3 +12,8 @@ def url(request):
|
|
|
11
12
|
if not value:
|
|
12
13
|
pytest.skip("需要通过 --url 参数传入链接")
|
|
13
14
|
return value
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@pytest.fixture
|
|
18
|
+
def proxy(request):
|
|
19
|
+
return request.config.getoption("--proxy")
|
|
@@ -18,9 +18,9 @@ TEST_URLS = [
|
|
|
18
18
|
|
|
19
19
|
|
|
20
20
|
@pytest.mark.parametrize("url", TEST_URLS)
|
|
21
|
-
def test_parse_sync(url: str) -> None:
|
|
21
|
+
def test_parse_sync(url: str, proxy: str | None) -> None:
|
|
22
22
|
try:
|
|
23
|
-
result = parse(url)
|
|
23
|
+
result = parse(url, proxy=proxy)
|
|
24
24
|
except WeChatVerifyError:
|
|
25
25
|
pytest.skip("WeChat returned verification page (IP rate-limited)")
|
|
26
26
|
return
|
|
@@ -29,9 +29,9 @@ def test_parse_sync(url: str) -> None:
|
|
|
29
29
|
|
|
30
30
|
@pytest.mark.parametrize("url", TEST_URLS)
|
|
31
31
|
@pytest.mark.asyncio
|
|
32
|
-
async def test_parse_async(url: str) -> None:
|
|
32
|
+
async def test_parse_async(url: str, proxy: str | None) -> None:
|
|
33
33
|
try:
|
|
34
|
-
result = await parse_async(url)
|
|
34
|
+
result = await parse_async(url, proxy=proxy)
|
|
35
35
|
except WeChatVerifyError:
|
|
36
36
|
pytest.skip("WeChat returned verification page (IP rate-limited)")
|
|
37
37
|
return
|
|
@@ -89,12 +89,12 @@ def _assert_result(result: ArticleResult, url: str) -> None:
|
|
|
89
89
|
assert result.is_valid
|
|
90
90
|
|
|
91
91
|
|
|
92
|
-
def test_fetch_all(url: str) -> None:
|
|
92
|
+
def test_fetch_all(url: str, proxy: str | None) -> None:
|
|
93
93
|
"""抓取指定 URL 并打印所有采集到的参数。
|
|
94
94
|
|
|
95
|
-
用法: pytest tests/test_parser.py::test_fetch_all -s --url <URL>
|
|
95
|
+
用法: pytest tests/test_parser.py::test_fetch_all -s --url <URL> [--proxy <PROXY>]
|
|
96
96
|
"""
|
|
97
|
-
result = parse(url)
|
|
97
|
+
result = parse(url, proxy=proxy)
|
|
98
98
|
print(f"\n{'='*60}")
|
|
99
99
|
print(f"URL: {url}")
|
|
100
100
|
print(f"公众号ID(B64):{result.mp_id_b64}")
|
|
@@ -120,10 +120,24 @@ def test_fetch_all(url: str) -> None:
|
|
|
120
120
|
print(f"{'='*60}")
|
|
121
121
|
|
|
122
122
|
|
|
123
|
-
def test_fetch_markdown(url: str) -> None:
|
|
123
|
+
def test_fetch_markdown(url: str, proxy: str | None) -> None:
|
|
124
124
|
"""抓取指定 URL 并只打印 Markdown 内容。
|
|
125
125
|
|
|
126
|
-
用法: pytest tests/test_parser.py::test_fetch_markdown -s --url <URL>
|
|
126
|
+
用法: pytest tests/test_parser.py::test_fetch_markdown -s --url <URL> [--proxy <PROXY>]
|
|
127
127
|
"""
|
|
128
|
-
result = parse(url)
|
|
128
|
+
result = parse(url, proxy=proxy)
|
|
129
129
|
print(f"\n{result.article_markdown}")
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def test_fetch_raw_html(url: str, proxy: str | None) -> None:
|
|
133
|
+
"""抓取指定 URL 并打印原始 HTML 网页源代码。
|
|
134
|
+
|
|
135
|
+
用法: pytest tests/test_parser.py::test_fetch_raw_html -s --url <URL> [--proxy <PROXY>]
|
|
136
|
+
"""
|
|
137
|
+
result = parse(url, proxy=proxy, include_raw_html=True)
|
|
138
|
+
assert result.raw_html, "raw_html should not be empty when include_raw_html=True"
|
|
139
|
+
print(f"\n{'='*60}")
|
|
140
|
+
print(f"URL: {url}")
|
|
141
|
+
print(f"raw_html长度: {len(result.raw_html)}")
|
|
142
|
+
print(f"{'='*60}")
|
|
143
|
+
print(result.raw_html)
|
|
File without changes
|
|
File without changes
|
{wechat_article_parser-0.0.3 → wechat_article_parser-0.0.5}/src/wechat_article_parser/__init__.py
RENAMED
|
File without changes
|