wechat-article-parser 0.0.5__tar.gz → 0.0.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: wechat-article-parser
3
- Version: 0.0.5
3
+ Version: 0.0.6
4
4
  Summary: WeChat MP article parser - extract metadata and content from WeChat public account articles
5
5
  License-Expression: MIT
6
6
  License-File: LICENSE
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "wechat-article-parser"
7
- version = "0.0.5"
7
+ version = "0.0.6"
8
8
  description = "WeChat MP article parser - extract metadata and content from WeChat public account articles"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -38,6 +38,15 @@ def _to_markdown(html: str) -> str:
38
38
  return _MarkdownConverter().convert(html)
39
39
 
40
40
 
41
+ def _to_markdown_plain(html: str) -> str:
42
+ """用于纯文本类内容(纯文本/转载/视频/轮播/全屏)的转换。
43
+
44
+ 这些内容由 <br> 承载换行,必须保留;故使用未覆写 convert_p 的基础转换器,
45
+ 避免 _MarkdownConverter 折叠空白时把换行一并吞掉。
46
+ """
47
+ return MarkdownConverter().convert(html)
48
+
49
+
41
50
  # ---------------------------------------------------------------------------
42
51
  # 文本解码辅助函数
43
52
  # ---------------------------------------------------------------------------
@@ -317,7 +326,7 @@ def _extract_repost_content(content_tag: Tag, result: ArticleResult) -> None:
317
326
  if href:
318
327
  html_content += f'<p><a href="{href}">查看原文</a></p>'
319
328
 
320
- result.article_markdown = _to_markdown(html_content)
329
+ result.article_markdown = _to_markdown_plain(html_content)
321
330
 
322
331
 
323
332
  def _extract_plain_text_content(soup: BeautifulSoup, result: ArticleResult) -> None:
@@ -331,7 +340,7 @@ def _extract_plain_text_content(soup: BeautifulSoup, result: ArticleResult) -> N
331
340
  if m:
332
341
  text = _decode_text(m.group(1), preserve_newlines=True)
333
342
  text = unquote(text)
334
- result.article_markdown = _to_markdown(f"<p>{text}</p>")
343
+ result.article_markdown = _to_markdown_plain(f"<p>{text}</p>")
335
344
 
336
345
 
337
346
  def _extract_swiper_content(soup: BeautifulSoup, result: ArticleResult) -> None:
@@ -350,7 +359,7 @@ def _extract_swiper_content(soup: BeautifulSoup, result: ArticleResult) -> None:
350
359
  html_parts.append(f"<p>{text}</p>")
351
360
 
352
361
  if html_parts:
353
- result.article_markdown = _to_markdown("".join(html_parts))
362
+ result.article_markdown = _to_markdown_plain("".join(html_parts))
354
363
 
355
364
 
356
365
  def _extract_fullscreen_content(soup: BeautifulSoup, result: ArticleResult) -> None:
@@ -371,9 +380,10 @@ def _extract_fullscreen_content(soup: BeautifulSoup, result: ArticleResult) -> N
371
380
  html_parts.append(f'<img src="{img}" /><br>')
372
381
 
373
382
  # 从 text_page_info.content_noencode 或 content 中提取文本
383
+ # 文本可能被 JsDecode('...') 包裹,也可能是裸的单引号字符串,两种都要兼容
374
384
  for field in ("content_noencode", "content"):
375
385
  m = re.search(
376
- rf"{field}:\s*JsDecode\('(.*?)'\)",
386
+ rf"{field}:\s*(?:JsDecode\(\s*)?'(.*?)'",
377
387
  script.text,
378
388
  re.DOTALL,
379
389
  )
@@ -384,7 +394,7 @@ def _extract_fullscreen_content(soup: BeautifulSoup, result: ArticleResult) -> N
384
394
  break
385
395
 
386
396
  if html_parts:
387
- result.article_markdown = _to_markdown("".join(html_parts))
397
+ result.article_markdown = _to_markdown_plain("".join(html_parts))
388
398
  return
389
399
 
390
400
 
@@ -14,6 +14,7 @@ TEST_URLS = [
14
14
  "https://mp.weixin.qq.com/s/sfwGsafriO9sm6PfbQVXhw",
15
15
  "https://mp.weixin.qq.com/s/h8E6riExCaH2Znmnprj-WQ",
16
16
  "https://mp.weixin.qq.com/s/ySQdtsRlRmAl_skdc5HQ-A",
17
+ "https://mp.weixin.qq.com/s/MnkArbYQNp3tF29gujMUnQ",
17
18
  ]
18
19
 
19
20