wechat-article-parser 0.0.4__tar.gz → 0.0.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {wechat_article_parser-0.0.4 → wechat_article_parser-0.0.5}/PKG-INFO +10 -1
- {wechat_article_parser-0.0.4 → wechat_article_parser-0.0.5}/README.md +9 -0
- {wechat_article_parser-0.0.4 → wechat_article_parser-0.0.5}/pyproject.toml +2 -1
- {wechat_article_parser-0.0.4 → wechat_article_parser-0.0.5}/src/wechat_article_parser/models.py +3 -0
- {wechat_article_parser-0.0.4 → wechat_article_parser-0.0.5}/src/wechat_article_parser/parser.py +12 -2
- {wechat_article_parser-0.0.4 → wechat_article_parser-0.0.5}/tests/test_parser.py +14 -0
- {wechat_article_parser-0.0.4 → wechat_article_parser-0.0.5}/.gitignore +0 -0
- {wechat_article_parser-0.0.4 → wechat_article_parser-0.0.5}/LICENSE +0 -0
- {wechat_article_parser-0.0.4 → wechat_article_parser-0.0.5}/src/wechat_article_parser/__init__.py +0 -0
- {wechat_article_parser-0.0.4 → wechat_article_parser-0.0.5}/tests/conftest.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: wechat-article-parser
|
|
3
|
-
Version: 0.0.
|
|
3
|
+
Version: 0.0.5
|
|
4
4
|
Summary: WeChat MP article parser - extract metadata and content from WeChat public account articles
|
|
5
5
|
License-Expression: MIT
|
|
6
6
|
License-File: LICENSE
|
|
@@ -76,6 +76,7 @@ print(result.article_markdown)
|
|
|
76
76
|
- `timeout`:请求超时时间,单位秒,默认 15
|
|
77
77
|
- `user_agent`:自定义 User-Agent,不传则使用内置默认值
|
|
78
78
|
- `proxy`:HTTP/HTTPS 代理地址,不传则直连
|
|
79
|
+
- `include_raw_html`:是否在返回结果中带上原始 HTML 网页源代码,默认 `False`
|
|
79
80
|
|
|
80
81
|
```python
|
|
81
82
|
result = parse(
|
|
@@ -83,6 +84,7 @@ result = parse(
|
|
|
83
84
|
timeout=30,
|
|
84
85
|
user_agent="MyBot/1.0",
|
|
85
86
|
proxy="http://user:pass@127.0.0.1:7890",
|
|
87
|
+
include_raw_html=True,
|
|
86
88
|
)
|
|
87
89
|
```
|
|
88
90
|
|
|
@@ -108,6 +110,7 @@ result = parse(
|
|
|
108
110
|
| `article_description` | `str` | 文章摘要 |
|
|
109
111
|
| `article_markdown` | `str` | 文章正文的 Markdown 内容 |
|
|
110
112
|
| `article_publish_time` | `int` | 发布时间(Unix 时间戳) |
|
|
113
|
+
| `raw_html` | `str` | 原始 HTML 网页源代码(仅在 `include_raw_html=True` 时填充,否则为空字符串) |
|
|
111
114
|
| `images` | `list[str]` | 文章中提取的所有图片链接 |
|
|
112
115
|
| `is_valid` | `bool` | 关键字段是否全部解析成功(属性) |
|
|
113
116
|
|
|
@@ -206,6 +209,12 @@ pytest tests/test_parser.py::test_fetch_all -s --url "https://mp.weixin.qq.com/s
|
|
|
206
209
|
pytest tests/test_parser.py::test_fetch_markdown -s --url "https://mp.weixin.qq.com/s/xxxxx"
|
|
207
210
|
```
|
|
208
211
|
|
|
212
|
+
### 测试单个链接 - 打印原始 HTML 网页源代码
|
|
213
|
+
|
|
214
|
+
```bash
|
|
215
|
+
pytest tests/test_parser.py::test_fetch_raw_html -s --url "https://mp.weixin.qq.com/s/xxxxx"
|
|
216
|
+
```
|
|
217
|
+
|
|
209
218
|
### 通过 HTTP 代理运行测试
|
|
210
219
|
|
|
211
220
|
所有测试命令都支持 `--proxy` 参数,传入后所有请求都会经由该代理;不传则直连。常用于 IP 被微信限流时换出口:
|
|
@@ -61,6 +61,7 @@ print(result.article_markdown)
|
|
|
61
61
|
- `timeout`:请求超时时间,单位秒,默认 15
|
|
62
62
|
- `user_agent`:自定义 User-Agent,不传则使用内置默认值
|
|
63
63
|
- `proxy`:HTTP/HTTPS 代理地址,不传则直连
|
|
64
|
+
- `include_raw_html`:是否在返回结果中带上原始 HTML 网页源代码,默认 `False`
|
|
64
65
|
|
|
65
66
|
```python
|
|
66
67
|
result = parse(
|
|
@@ -68,6 +69,7 @@ result = parse(
|
|
|
68
69
|
timeout=30,
|
|
69
70
|
user_agent="MyBot/1.0",
|
|
70
71
|
proxy="http://user:pass@127.0.0.1:7890",
|
|
72
|
+
include_raw_html=True,
|
|
71
73
|
)
|
|
72
74
|
```
|
|
73
75
|
|
|
@@ -93,6 +95,7 @@ result = parse(
|
|
|
93
95
|
| `article_description` | `str` | 文章摘要 |
|
|
94
96
|
| `article_markdown` | `str` | 文章正文的 Markdown 内容 |
|
|
95
97
|
| `article_publish_time` | `int` | 发布时间(Unix 时间戳) |
|
|
98
|
+
| `raw_html` | `str` | 原始 HTML 网页源代码(仅在 `include_raw_html=True` 时填充,否则为空字符串) |
|
|
96
99
|
| `images` | `list[str]` | 文章中提取的所有图片链接 |
|
|
97
100
|
| `is_valid` | `bool` | 关键字段是否全部解析成功(属性) |
|
|
98
101
|
|
|
@@ -191,6 +194,12 @@ pytest tests/test_parser.py::test_fetch_all -s --url "https://mp.weixin.qq.com/s
|
|
|
191
194
|
pytest tests/test_parser.py::test_fetch_markdown -s --url "https://mp.weixin.qq.com/s/xxxxx"
|
|
192
195
|
```
|
|
193
196
|
|
|
197
|
+
### 测试单个链接 - 打印原始 HTML 网页源代码
|
|
198
|
+
|
|
199
|
+
```bash
|
|
200
|
+
pytest tests/test_parser.py::test_fetch_raw_html -s --url "https://mp.weixin.qq.com/s/xxxxx"
|
|
201
|
+
```
|
|
202
|
+
|
|
194
203
|
### 通过 HTTP 代理运行测试
|
|
195
204
|
|
|
196
205
|
所有测试命令都支持 `--proxy` 参数,传入后所有请求都会经由该代理;不传则直连。常用于 IP 被微信限流时换出口:
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "wechat-article-parser"
|
|
7
|
-
version = "0.0.
|
|
7
|
+
version = "0.0.5"
|
|
8
8
|
description = "WeChat MP article parser - extract metadata and content from WeChat public account articles"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -26,3 +26,4 @@ packages = ["src/wechat_article_parser"]
|
|
|
26
26
|
|
|
27
27
|
[tool.pytest.ini_options]
|
|
28
28
|
asyncio_mode = "auto"
|
|
29
|
+
pythonpath = ["src"]
|
{wechat_article_parser-0.0.4 → wechat_article_parser-0.0.5}/src/wechat_article_parser/parser.py
RENAMED
|
@@ -482,6 +482,7 @@ def parse(
|
|
|
482
482
|
timeout: int = _TIMEOUT,
|
|
483
483
|
user_agent: str | None = None,
|
|
484
484
|
proxy: str | None = None,
|
|
485
|
+
include_raw_html: bool = False,
|
|
485
486
|
) -> ArticleResult:
|
|
486
487
|
"""抓取并解析微信公众号文章(同步方式)。
|
|
487
488
|
|
|
@@ -490,6 +491,7 @@ def parse(
|
|
|
490
491
|
timeout: 请求超时时间(秒)。
|
|
491
492
|
user_agent: 自定义 User-Agent,不传则使用内置默认值。
|
|
492
493
|
proxy: HTTP/HTTPS 代理地址,例如 "http://user:pass@host:port",不传则直连。
|
|
494
|
+
include_raw_html: 是否在结果中返回原始 HTML 网页源代码,默认不返回。
|
|
493
495
|
|
|
494
496
|
Returns:
|
|
495
497
|
包含解析数据的 ArticleResult。
|
|
@@ -503,7 +505,10 @@ def parse(
|
|
|
503
505
|
) as client:
|
|
504
506
|
response = client.get(url)
|
|
505
507
|
response.raise_for_status()
|
|
506
|
-
|
|
508
|
+
result = _parse_html(url, response.text)
|
|
509
|
+
if include_raw_html:
|
|
510
|
+
result.raw_html = response.text
|
|
511
|
+
return result
|
|
507
512
|
|
|
508
513
|
|
|
509
514
|
async def parse_async(
|
|
@@ -512,6 +517,7 @@ async def parse_async(
|
|
|
512
517
|
timeout: int = _TIMEOUT,
|
|
513
518
|
user_agent: str | None = None,
|
|
514
519
|
proxy: str | None = None,
|
|
520
|
+
include_raw_html: bool = False,
|
|
515
521
|
) -> ArticleResult:
|
|
516
522
|
"""抓取并解析微信公众号文章(异步方式)。
|
|
517
523
|
|
|
@@ -520,6 +526,7 @@ async def parse_async(
|
|
|
520
526
|
timeout: 请求超时时间(秒)。
|
|
521
527
|
user_agent: 自定义 User-Agent,不传则使用内置默认值。
|
|
522
528
|
proxy: HTTP/HTTPS 代理地址,例如 "http://user:pass@host:port",不传则直连。
|
|
529
|
+
include_raw_html: 是否在结果中返回原始 HTML 网页源代码,默认不返回。
|
|
523
530
|
|
|
524
531
|
Returns:
|
|
525
532
|
包含解析数据的 ArticleResult。
|
|
@@ -533,4 +540,7 @@ async def parse_async(
|
|
|
533
540
|
) as client:
|
|
534
541
|
response = await client.get(url)
|
|
535
542
|
response.raise_for_status()
|
|
536
|
-
|
|
543
|
+
result = _parse_html(url, response.text)
|
|
544
|
+
if include_raw_html:
|
|
545
|
+
result.raw_html = response.text
|
|
546
|
+
return result
|
|
@@ -127,3 +127,17 @@ def test_fetch_markdown(url: str, proxy: str | None) -> None:
|
|
|
127
127
|
"""
|
|
128
128
|
result = parse(url, proxy=proxy)
|
|
129
129
|
print(f"\n{result.article_markdown}")
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def test_fetch_raw_html(url: str, proxy: str | None) -> None:
|
|
133
|
+
"""抓取指定 URL 并打印原始 HTML 网页源代码。
|
|
134
|
+
|
|
135
|
+
用法: pytest tests/test_parser.py::test_fetch_raw_html -s --url <URL> [--proxy <PROXY>]
|
|
136
|
+
"""
|
|
137
|
+
result = parse(url, proxy=proxy, include_raw_html=True)
|
|
138
|
+
assert result.raw_html, "raw_html should not be empty when include_raw_html=True"
|
|
139
|
+
print(f"\n{'='*60}")
|
|
140
|
+
print(f"URL: {url}")
|
|
141
|
+
print(f"raw_html长度: {len(result.raw_html)}")
|
|
142
|
+
print(f"{'='*60}")
|
|
143
|
+
print(result.raw_html)
|
|
File without changes
|
|
File without changes
|
{wechat_article_parser-0.0.4 → wechat_article_parser-0.0.5}/src/wechat_article_parser/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|