fastfeedparser 0.5.9__tar.gz → 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.9
3
+ Version: 0.6.0
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = fastfeedparser
3
- version = 0.5.9
3
+ version = 0.6.0
4
4
  author = Vladimir Prelovac
5
5
  author_email = vlad@kagi.com
6
6
  description = High performance RSS, Atom, JSON and RDF feed parser in Python
@@ -1,4 +1,4 @@
1
1
  from .main import parse, FastFeedParserDict
2
2
 
3
- __version__ = "0.5.1"
3
+ __version__ = "0.6.0"
4
4
  __all__ = ["parse", "FastFeedParserDict"]
@@ -635,6 +635,7 @@ def _raise_for_non_feed_root(
635
635
 
636
636
 
637
637
  _RE_META_REFRESH_URL = re.compile(r'url\s*=\s*["\']?\s*([^"\'>\s]+)', re.IGNORECASE)
638
+ _MAX_META_REDIRECTS = 3
638
639
 
639
640
 
640
641
  def _extract_meta_refresh_url(content: str | bytes, base_url: str) -> str | None:
@@ -866,31 +867,32 @@ def parse(
866
867
  else:
867
868
  content = source
868
869
 
869
- try:
870
- return _parse_content(
871
- content,
872
- include_content=include_content,
873
- include_tags=include_tags,
874
- include_media=include_media,
875
- include_enclosures=include_enclosures,
876
- )
877
- except ValueError as e:
878
- if not is_url:
879
- raise
880
- assert isinstance(source, str)
881
- err_msg = str(e)
882
- if "HTML" not in err_msg and "not a valid RSS/Atom feed" not in err_msg:
883
- raise
884
- redirect_url = _extract_meta_refresh_url(content, source)
885
- if redirect_url is None:
886
- raise
887
- return parse(
888
- redirect_url,
889
- include_content=include_content,
890
- include_tags=include_tags,
891
- include_media=include_media,
892
- include_enclosures=include_enclosures,
893
- )
870
+ parse_kwargs = dict(
871
+ include_content=include_content,
872
+ include_tags=include_tags,
873
+ include_media=include_media,
874
+ include_enclosures=include_enclosures,
875
+ )
876
+
877
+ redirects_left = _MAX_META_REDIRECTS
878
+ while True:
879
+ try:
880
+ return _parse_content(content, **parse_kwargs)
881
+ except ValueError as e:
882
+ if not is_url:
883
+ raise
884
+ assert isinstance(source, str)
885
+ err_msg = str(e)
886
+ if "HTML" not in err_msg and "not a valid RSS/Atom feed" not in err_msg:
887
+ raise
888
+ if redirects_left <= 0:
889
+ raise ValueError("too many meta-refresh redirects") from e
890
+ redirect_url = _extract_meta_refresh_url(content, source)
891
+ if redirect_url is None:
892
+ raise
893
+ content = _fetch_url_content(redirect_url)
894
+ source = redirect_url
895
+ redirects_left -= 1
894
896
 
895
897
 
896
898
  def _parse_feed_info(
@@ -1051,6 +1053,35 @@ def _parse_feed_info(
1051
1053
  if managing_editor:
1052
1054
  feed["author"] = managing_editor
1053
1055
 
1056
+ # Parse feed-level image/icon/logo
1057
+ if feed_type == "atom":
1058
+ icon_el = channel.find(f"{{{atom_ns}}}icon")
1059
+ if icon_el is not None and icon_el.text:
1060
+ feed["icon"] = icon_el.text.strip()
1061
+ logo_el = channel.find(f"{{{atom_ns}}}logo")
1062
+ if logo_el is not None and logo_el.text:
1063
+ feed["logo"] = logo_el.text.strip()
1064
+ elif feed_type == "rss":
1065
+ image_el = channel.find("image")
1066
+ if image_el is not None:
1067
+ image: dict[str, Optional[str]] = {}
1068
+ for sub_tag in ("url", "title", "link"):
1069
+ sub_el = image_el.find(sub_tag)
1070
+ if sub_el is not None and sub_el.text:
1071
+ image[sub_tag] = sub_el.text.strip()
1072
+ if image.get("url"):
1073
+ feed["image"] = image
1074
+ elif feed_type == "rdf":
1075
+ rdf_image_el = channel.find("{http://purl.org/rss/1.0/}image")
1076
+ if rdf_image_el is not None:
1077
+ image = {}
1078
+ for sub_tag in ("title", "link", "url"):
1079
+ sub_el = rdf_image_el.find(f"{{http://purl.org/rss/1.0/}}{sub_tag}")
1080
+ if sub_el is not None and sub_el.text:
1081
+ image[sub_tag] = sub_el.text.strip()
1082
+ if image.get("url"):
1083
+ feed["image"] = image
1084
+
1054
1085
  # Parse feed-level tags/categories
1055
1086
  if include_tags:
1056
1087
  tags = _parse_tags(channel, feed_type, atom_ns)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.9
3
+ Version: 0.6.0
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -10,4 +10,5 @@ src/fastfeedparser.egg-info/dependency_links.txt
10
10
  src/fastfeedparser.egg-info/requires.txt
11
11
  src/fastfeedparser.egg-info/top_level.txt
12
12
  tests/test_encoding.py
13
+ tests/test_feed_image.py
13
14
  tests/test_integration.py
@@ -0,0 +1,136 @@
1
+ """Tests for feed-level image/icon/logo extraction."""
2
+
3
+ from fastfeedparser import parse
4
+
5
+
6
+ def test_rss_feed_image():
7
+ xml = b"""<?xml version="1.0" encoding="UTF-8"?>
8
+ <rss version="2.0">
9
+ <channel>
10
+ <title>Test Feed</title>
11
+ <link>https://example.com</link>
12
+ <image>
13
+ <url>https://example.com/logo.png</url>
14
+ <title>Test Feed</title>
15
+ <link>https://example.com</link>
16
+ </image>
17
+ <item>
18
+ <title>Entry 1</title>
19
+ </item>
20
+ </channel>
21
+ </rss>"""
22
+ result = parse(xml)
23
+ assert result["feed"]["image"] == {
24
+ "url": "https://example.com/logo.png",
25
+ "title": "Test Feed",
26
+ "link": "https://example.com",
27
+ }
28
+
29
+
30
+ def test_rss_feed_image_url_only():
31
+ xml = b"""<?xml version="1.0" encoding="UTF-8"?>
32
+ <rss version="2.0">
33
+ <channel>
34
+ <title>Test</title>
35
+ <image>
36
+ <url>https://example.com/icon.png</url>
37
+ </image>
38
+ <item><title>Entry</title></item>
39
+ </channel>
40
+ </rss>"""
41
+ result = parse(xml)
42
+ assert result["feed"]["image"]["url"] == "https://example.com/icon.png"
43
+ assert "title" not in result["feed"]["image"]
44
+
45
+
46
+ def test_rss_feed_no_image():
47
+ xml = b"""<?xml version="1.0" encoding="UTF-8"?>
48
+ <rss version="2.0">
49
+ <channel>
50
+ <title>Test</title>
51
+ <item><title>Entry</title></item>
52
+ </channel>
53
+ </rss>"""
54
+ result = parse(xml)
55
+ assert "image" not in result["feed"]
56
+
57
+
58
+ def test_rss_feed_image_empty_url():
59
+ xml = b"""<?xml version="1.0" encoding="UTF-8"?>
60
+ <rss version="2.0">
61
+ <channel>
62
+ <title>Test</title>
63
+ <image>
64
+ <url></url>
65
+ <title>Test</title>
66
+ </image>
67
+ <item><title>Entry</title></item>
68
+ </channel>
69
+ </rss>"""
70
+ result = parse(xml)
71
+ assert "image" not in result["feed"]
72
+
73
+
74
+ def test_atom_feed_icon_and_logo():
75
+ xml = b"""<?xml version="1.0" encoding="UTF-8"?>
76
+ <feed xmlns="http://www.w3.org/2005/Atom">
77
+ <title>Test Feed</title>
78
+ <icon>https://example.com/favicon.ico</icon>
79
+ <logo>https://example.com/banner.png</logo>
80
+ <entry>
81
+ <title>Entry 1</title>
82
+ <id>urn:entry:1</id>
83
+ </entry>
84
+ </feed>"""
85
+ result = parse(xml)
86
+ assert result["feed"]["icon"] == "https://example.com/favicon.ico"
87
+ assert result["feed"]["logo"] == "https://example.com/banner.png"
88
+
89
+
90
+ def test_atom_feed_icon_only():
91
+ xml = b"""<?xml version="1.0" encoding="UTF-8"?>
92
+ <feed xmlns="http://www.w3.org/2005/Atom">
93
+ <title>Test</title>
94
+ <icon>https://example.com/favicon.ico</icon>
95
+ <entry><title>E</title><id>urn:1</id></entry>
96
+ </feed>"""
97
+ result = parse(xml)
98
+ assert result["feed"]["icon"] == "https://example.com/favicon.ico"
99
+ assert "logo" not in result["feed"]
100
+
101
+
102
+ def test_atom_feed_no_icon():
103
+ xml = b"""<?xml version="1.0" encoding="UTF-8"?>
104
+ <feed xmlns="http://www.w3.org/2005/Atom">
105
+ <title>Test</title>
106
+ <entry><title>E</title><id>urn:1</id></entry>
107
+ </feed>"""
108
+ result = parse(xml)
109
+ assert "icon" not in result["feed"]
110
+ assert "logo" not in result["feed"]
111
+
112
+
113
+ def test_rdf_feed_image():
114
+ xml = b"""<?xml version="1.0" encoding="UTF-8"?>
115
+ <rdf:RDF
116
+ xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"
117
+ xmlns="http://purl.org/rss/1.0/">
118
+ <channel>
119
+ <title>Test RDF Feed</title>
120
+ <link>https://example.com</link>
121
+ </channel>
122
+ <image>
123
+ <url>https://example.com/rdf-logo.png</url>
124
+ <title>RDF Feed</title>
125
+ <link>https://example.com</link>
126
+ </image>
127
+ <item>
128
+ <title>Entry 1</title>
129
+ </item>
130
+ </rdf:RDF>"""
131
+ result = parse(xml)
132
+ assert result["feed"]["image"] == {
133
+ "url": "https://example.com/rdf-logo.png",
134
+ "title": "RDF Feed",
135
+ "link": "https://example.com",
136
+ }
File without changes
File without changes