wechat-article-parser 0.0.4__tar.gz → 0.0.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: wechat-article-parser
3
- Version: 0.0.4
3
+ Version: 0.0.5
4
4
  Summary: WeChat MP article parser - extract metadata and content from WeChat public account articles
5
5
  License-Expression: MIT
6
6
  License-File: LICENSE
@@ -76,6 +76,7 @@ print(result.article_markdown)
76
76
  - `timeout`:请求超时时间,单位秒,默认 15
77
77
  - `user_agent`:自定义 User-Agent,不传则使用内置默认值
78
78
  - `proxy`:HTTP/HTTPS 代理地址,不传则直连
79
+ - `include_raw_html`:是否在返回结果中带上原始 HTML 网页源代码,默认 `False`
79
80
 
80
81
  ```python
81
82
  result = parse(
@@ -83,6 +84,7 @@ result = parse(
83
84
  timeout=30,
84
85
  user_agent="MyBot/1.0",
85
86
  proxy="http://user:pass@127.0.0.1:7890",
87
+ include_raw_html=True,
86
88
  )
87
89
  ```
88
90
 
@@ -108,6 +110,7 @@ result = parse(
108
110
  | `article_description` | `str` | 文章摘要 |
109
111
  | `article_markdown` | `str` | 文章正文的 Markdown 内容 |
110
112
  | `article_publish_time` | `int` | 发布时间(Unix 时间戳) |
113
+ | `raw_html` | `str` | 原始 HTML 网页源代码(仅在 `include_raw_html=True` 时填充,否则为空字符串) |
111
114
  | `images` | `list[str]` | 文章中提取的所有图片链接 |
112
115
  | `is_valid` | `bool` | 关键字段是否全部解析成功(属性) |
113
116
 
@@ -206,6 +209,12 @@ pytest tests/test_parser.py::test_fetch_all -s --url "https://mp.weixin.qq.com/s
206
209
  pytest tests/test_parser.py::test_fetch_markdown -s --url "https://mp.weixin.qq.com/s/xxxxx"
207
210
  ```
208
211
 
212
+ ### 测试单个链接 - 打印原始 HTML 网页源代码
213
+
214
+ ```bash
215
+ pytest tests/test_parser.py::test_fetch_raw_html -s --url "https://mp.weixin.qq.com/s/xxxxx"
216
+ ```
217
+
209
218
  ### 通过 HTTP 代理运行测试
210
219
 
211
220
  所有测试命令都支持 `--proxy` 参数,传入后所有请求都会经由该代理;不传则直连。常用于 IP 被微信限流时换出口:
@@ -61,6 +61,7 @@ print(result.article_markdown)
61
61
  - `timeout`:请求超时时间,单位秒,默认 15
62
62
  - `user_agent`:自定义 User-Agent,不传则使用内置默认值
63
63
  - `proxy`:HTTP/HTTPS 代理地址,不传则直连
64
+ - `include_raw_html`:是否在返回结果中带上原始 HTML 网页源代码,默认 `False`
64
65
 
65
66
  ```python
66
67
  result = parse(
@@ -68,6 +69,7 @@ result = parse(
68
69
  timeout=30,
69
70
  user_agent="MyBot/1.0",
70
71
  proxy="http://user:pass@127.0.0.1:7890",
72
+ include_raw_html=True,
71
73
  )
72
74
  ```
73
75
 
@@ -93,6 +95,7 @@ result = parse(
93
95
  | `article_description` | `str` | 文章摘要 |
94
96
  | `article_markdown` | `str` | 文章正文的 Markdown 内容 |
95
97
  | `article_publish_time` | `int` | 发布时间(Unix 时间戳) |
98
+ | `raw_html` | `str` | 原始 HTML 网页源代码(仅在 `include_raw_html=True` 时填充,否则为空字符串) |
96
99
  | `images` | `list[str]` | 文章中提取的所有图片链接 |
97
100
  | `is_valid` | `bool` | 关键字段是否全部解析成功(属性) |
98
101
 
@@ -191,6 +194,12 @@ pytest tests/test_parser.py::test_fetch_all -s --url "https://mp.weixin.qq.com/s
191
194
  pytest tests/test_parser.py::test_fetch_markdown -s --url "https://mp.weixin.qq.com/s/xxxxx"
192
195
  ```
193
196
 
197
+ ### 测试单个链接 - 打印原始 HTML 网页源代码
198
+
199
+ ```bash
200
+ pytest tests/test_parser.py::test_fetch_raw_html -s --url "https://mp.weixin.qq.com/s/xxxxx"
201
+ ```
202
+
194
203
  ### 通过 HTTP 代理运行测试
195
204
 
196
205
  所有测试命令都支持 `--proxy` 参数,传入后所有请求都会经由该代理;不传则直连。常用于 IP 被微信限流时换出口:
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "wechat-article-parser"
7
- version = "0.0.4"
7
+ version = "0.0.5"
8
8
  description = "WeChat MP article parser - extract metadata and content from WeChat public account articles"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -26,3 +26,4 @@ packages = ["src/wechat_article_parser"]
26
26
 
27
27
  [tool.pytest.ini_options]
28
28
  asyncio_mode = "auto"
29
+ pythonpath = ["src"]
@@ -46,6 +46,9 @@ class ArticleResult:
46
46
  article_markdown: str = ""
47
47
  article_publish_time: int = 0
48
48
 
49
+ # 原始 HTML 网页源代码(仅在调用方显式要求时才会填充)
50
+ raw_html: str = ""
51
+
49
52
  # 文章中提取的图片列表
50
53
  images: list[str] = field(default_factory=list)
51
54
 
@@ -482,6 +482,7 @@ def parse(
482
482
  timeout: int = _TIMEOUT,
483
483
  user_agent: str | None = None,
484
484
  proxy: str | None = None,
485
+ include_raw_html: bool = False,
485
486
  ) -> ArticleResult:
486
487
  """抓取并解析微信公众号文章(同步方式)。
487
488
 
@@ -490,6 +491,7 @@ def parse(
490
491
  timeout: 请求超时时间(秒)。
491
492
  user_agent: 自定义 User-Agent,不传则使用内置默认值。
492
493
  proxy: HTTP/HTTPS 代理地址,例如 "http://user:pass@host:port",不传则直连。
494
+ include_raw_html: 是否在结果中返回原始 HTML 网页源代码,默认不返回。
493
495
 
494
496
  Returns:
495
497
  包含解析数据的 ArticleResult。
@@ -503,7 +505,10 @@ def parse(
503
505
  ) as client:
504
506
  response = client.get(url)
505
507
  response.raise_for_status()
506
- return _parse_html(url, response.text)
508
+ result = _parse_html(url, response.text)
509
+ if include_raw_html:
510
+ result.raw_html = response.text
511
+ return result
507
512
 
508
513
 
509
514
  async def parse_async(
@@ -512,6 +517,7 @@ async def parse_async(
512
517
  timeout: int = _TIMEOUT,
513
518
  user_agent: str | None = None,
514
519
  proxy: str | None = None,
520
+ include_raw_html: bool = False,
515
521
  ) -> ArticleResult:
516
522
  """抓取并解析微信公众号文章(异步方式)。
517
523
 
@@ -520,6 +526,7 @@ async def parse_async(
520
526
  timeout: 请求超时时间(秒)。
521
527
  user_agent: 自定义 User-Agent,不传则使用内置默认值。
522
528
  proxy: HTTP/HTTPS 代理地址,例如 "http://user:pass@host:port",不传则直连。
529
+ include_raw_html: 是否在结果中返回原始 HTML 网页源代码,默认不返回。
523
530
 
524
531
  Returns:
525
532
  包含解析数据的 ArticleResult。
@@ -533,4 +540,7 @@ async def parse_async(
533
540
  ) as client:
534
541
  response = await client.get(url)
535
542
  response.raise_for_status()
536
- return _parse_html(url, response.text)
543
+ result = _parse_html(url, response.text)
544
+ if include_raw_html:
545
+ result.raw_html = response.text
546
+ return result
@@ -127,3 +127,17 @@ def test_fetch_markdown(url: str, proxy: str | None) -> None:
127
127
  """
128
128
  result = parse(url, proxy=proxy)
129
129
  print(f"\n{result.article_markdown}")
130
+
131
+
132
+ def test_fetch_raw_html(url: str, proxy: str | None) -> None:
133
+ """抓取指定 URL 并打印原始 HTML 网页源代码。
134
+
135
+ 用法: pytest tests/test_parser.py::test_fetch_raw_html -s --url <URL> [--proxy <PROXY>]
136
+ """
137
+ result = parse(url, proxy=proxy, include_raw_html=True)
138
+ assert result.raw_html, "raw_html should not be empty when include_raw_html=True"
139
+ print(f"\n{'='*60}")
140
+ print(f"URL: {url}")
141
+ print(f"raw_html长度: {len(result.raw_html)}")
142
+ print(f"{'='*60}")
143
+ print(result.raw_html)