wechat-article-parser 0.0.5__tar.gz → 0.0.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {wechat_article_parser-0.0.5 → wechat_article_parser-0.0.6}/PKG-INFO +1 -1
- {wechat_article_parser-0.0.5 → wechat_article_parser-0.0.6}/pyproject.toml +1 -1
- {wechat_article_parser-0.0.5 → wechat_article_parser-0.0.6}/src/wechat_article_parser/parser.py +15 -5
- {wechat_article_parser-0.0.5 → wechat_article_parser-0.0.6}/tests/test_parser.py +1 -0
- {wechat_article_parser-0.0.5 → wechat_article_parser-0.0.6}/.gitignore +0 -0
- {wechat_article_parser-0.0.5 → wechat_article_parser-0.0.6}/LICENSE +0 -0
- {wechat_article_parser-0.0.5 → wechat_article_parser-0.0.6}/README.md +0 -0
- {wechat_article_parser-0.0.5 → wechat_article_parser-0.0.6}/src/wechat_article_parser/__init__.py +0 -0
- {wechat_article_parser-0.0.5 → wechat_article_parser-0.0.6}/src/wechat_article_parser/models.py +0 -0
- {wechat_article_parser-0.0.5 → wechat_article_parser-0.0.6}/tests/conftest.py +0 -0
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "wechat-article-parser"
|
|
7
|
-
version = "0.0.
|
|
7
|
+
version = "0.0.6"
|
|
8
8
|
description = "WeChat MP article parser - extract metadata and content from WeChat public account articles"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
{wechat_article_parser-0.0.5 → wechat_article_parser-0.0.6}/src/wechat_article_parser/parser.py
RENAMED
|
@@ -38,6 +38,15 @@ def _to_markdown(html: str) -> str:
|
|
|
38
38
|
return _MarkdownConverter().convert(html)
|
|
39
39
|
|
|
40
40
|
|
|
41
|
+
def _to_markdown_plain(html: str) -> str:
|
|
42
|
+
"""用于纯文本类内容(纯文本/转载/视频/轮播/全屏)的转换。
|
|
43
|
+
|
|
44
|
+
这些内容由 <br> 承载换行,必须保留;故使用未覆写 convert_p 的基础转换器,
|
|
45
|
+
避免 _MarkdownConverter 折叠空白时把换行一并吞掉。
|
|
46
|
+
"""
|
|
47
|
+
return MarkdownConverter().convert(html)
|
|
48
|
+
|
|
49
|
+
|
|
41
50
|
# ---------------------------------------------------------------------------
|
|
42
51
|
# 文本解码辅助函数
|
|
43
52
|
# ---------------------------------------------------------------------------
|
|
@@ -317,7 +326,7 @@ def _extract_repost_content(content_tag: Tag, result: ArticleResult) -> None:
|
|
|
317
326
|
if href:
|
|
318
327
|
html_content += f'<p><a href="{href}">查看原文</a></p>'
|
|
319
328
|
|
|
320
|
-
result.article_markdown =
|
|
329
|
+
result.article_markdown = _to_markdown_plain(html_content)
|
|
321
330
|
|
|
322
331
|
|
|
323
332
|
def _extract_plain_text_content(soup: BeautifulSoup, result: ArticleResult) -> None:
|
|
@@ -331,7 +340,7 @@ def _extract_plain_text_content(soup: BeautifulSoup, result: ArticleResult) -> N
|
|
|
331
340
|
if m:
|
|
332
341
|
text = _decode_text(m.group(1), preserve_newlines=True)
|
|
333
342
|
text = unquote(text)
|
|
334
|
-
result.article_markdown =
|
|
343
|
+
result.article_markdown = _to_markdown_plain(f"<p>{text}</p>")
|
|
335
344
|
|
|
336
345
|
|
|
337
346
|
def _extract_swiper_content(soup: BeautifulSoup, result: ArticleResult) -> None:
|
|
@@ -350,7 +359,7 @@ def _extract_swiper_content(soup: BeautifulSoup, result: ArticleResult) -> None:
|
|
|
350
359
|
html_parts.append(f"<p>{text}</p>")
|
|
351
360
|
|
|
352
361
|
if html_parts:
|
|
353
|
-
result.article_markdown =
|
|
362
|
+
result.article_markdown = _to_markdown_plain("".join(html_parts))
|
|
354
363
|
|
|
355
364
|
|
|
356
365
|
def _extract_fullscreen_content(soup: BeautifulSoup, result: ArticleResult) -> None:
|
|
@@ -371,9 +380,10 @@ def _extract_fullscreen_content(soup: BeautifulSoup, result: ArticleResult) -> N
|
|
|
371
380
|
html_parts.append(f'<img src="{img}" /><br>')
|
|
372
381
|
|
|
373
382
|
# 从 text_page_info.content_noencode 或 content 中提取文本
|
|
383
|
+
# 文本可能被 JsDecode('...') 包裹,也可能是裸的单引号字符串,两种都要兼容
|
|
374
384
|
for field in ("content_noencode", "content"):
|
|
375
385
|
m = re.search(
|
|
376
|
-
rf"{field}:\s*JsDecode\('(.*?)'
|
|
386
|
+
rf"{field}:\s*(?:JsDecode\(\s*)?'(.*?)'",
|
|
377
387
|
script.text,
|
|
378
388
|
re.DOTALL,
|
|
379
389
|
)
|
|
@@ -384,7 +394,7 @@ def _extract_fullscreen_content(soup: BeautifulSoup, result: ArticleResult) -> N
|
|
|
384
394
|
break
|
|
385
395
|
|
|
386
396
|
if html_parts:
|
|
387
|
-
result.article_markdown =
|
|
397
|
+
result.article_markdown = _to_markdown_plain("".join(html_parts))
|
|
388
398
|
return
|
|
389
399
|
|
|
390
400
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{wechat_article_parser-0.0.5 → wechat_article_parser-0.0.6}/src/wechat_article_parser/__init__.py
RENAMED
|
File without changes
|
{wechat_article_parser-0.0.5 → wechat_article_parser-0.0.6}/src/wechat_article_parser/models.py
RENAMED
|
File without changes
|
|
File without changes
|