fastfeedparser 0.5.9__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fastfeedparser-0.5.9/src/fastfeedparser.egg-info → fastfeedparser-0.6.0}/PKG-INFO +1 -1
- {fastfeedparser-0.5.9 → fastfeedparser-0.6.0}/setup.cfg +1 -1
- {fastfeedparser-0.5.9 → fastfeedparser-0.6.0}/src/fastfeedparser/__init__.py +1 -1
- {fastfeedparser-0.5.9 → fastfeedparser-0.6.0}/src/fastfeedparser/main.py +56 -25
- {fastfeedparser-0.5.9 → fastfeedparser-0.6.0/src/fastfeedparser.egg-info}/PKG-INFO +1 -1
- {fastfeedparser-0.5.9 → fastfeedparser-0.6.0}/src/fastfeedparser.egg-info/SOURCES.txt +1 -0
- fastfeedparser-0.6.0/tests/test_feed_image.py +136 -0
- {fastfeedparser-0.5.9 → fastfeedparser-0.6.0}/LICENSE +0 -0
- {fastfeedparser-0.5.9 → fastfeedparser-0.6.0}/README.md +0 -0
- {fastfeedparser-0.5.9 → fastfeedparser-0.6.0}/pyproject.toml +0 -0
- {fastfeedparser-0.5.9 → fastfeedparser-0.6.0}/src/fastfeedparser.egg-info/dependency_links.txt +0 -0
- {fastfeedparser-0.5.9 → fastfeedparser-0.6.0}/src/fastfeedparser.egg-info/requires.txt +0 -0
- {fastfeedparser-0.5.9 → fastfeedparser-0.6.0}/src/fastfeedparser.egg-info/top_level.txt +0 -0
- {fastfeedparser-0.5.9 → fastfeedparser-0.6.0}/tests/test_encoding.py +0 -0
- {fastfeedparser-0.5.9 → fastfeedparser-0.6.0}/tests/test_integration.py +0 -0
|
@@ -635,6 +635,7 @@ def _raise_for_non_feed_root(
|
|
|
635
635
|
|
|
636
636
|
|
|
637
637
|
_RE_META_REFRESH_URL = re.compile(r'url\s*=\s*["\']?\s*([^"\'>\s]+)', re.IGNORECASE)
|
|
638
|
+
_MAX_META_REDIRECTS = 3
|
|
638
639
|
|
|
639
640
|
|
|
640
641
|
def _extract_meta_refresh_url(content: str | bytes, base_url: str) -> str | None:
|
|
@@ -866,31 +867,32 @@ def parse(
|
|
|
866
867
|
else:
|
|
867
868
|
content = source
|
|
868
869
|
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
|
|
875
|
-
|
|
876
|
-
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
|
|
881
|
-
|
|
882
|
-
|
|
883
|
-
|
|
884
|
-
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
|
|
891
|
-
|
|
892
|
-
|
|
893
|
-
|
|
870
|
+
parse_kwargs = dict(
|
|
871
|
+
include_content=include_content,
|
|
872
|
+
include_tags=include_tags,
|
|
873
|
+
include_media=include_media,
|
|
874
|
+
include_enclosures=include_enclosures,
|
|
875
|
+
)
|
|
876
|
+
|
|
877
|
+
redirects_left = _MAX_META_REDIRECTS
|
|
878
|
+
while True:
|
|
879
|
+
try:
|
|
880
|
+
return _parse_content(content, **parse_kwargs)
|
|
881
|
+
except ValueError as e:
|
|
882
|
+
if not is_url:
|
|
883
|
+
raise
|
|
884
|
+
assert isinstance(source, str)
|
|
885
|
+
err_msg = str(e)
|
|
886
|
+
if "HTML" not in err_msg and "not a valid RSS/Atom feed" not in err_msg:
|
|
887
|
+
raise
|
|
888
|
+
if redirects_left <= 0:
|
|
889
|
+
raise ValueError("too many meta-refresh redirects") from e
|
|
890
|
+
redirect_url = _extract_meta_refresh_url(content, source)
|
|
891
|
+
if redirect_url is None:
|
|
892
|
+
raise
|
|
893
|
+
content = _fetch_url_content(redirect_url)
|
|
894
|
+
source = redirect_url
|
|
895
|
+
redirects_left -= 1
|
|
894
896
|
|
|
895
897
|
|
|
896
898
|
def _parse_feed_info(
|
|
@@ -1051,6 +1053,35 @@ def _parse_feed_info(
|
|
|
1051
1053
|
if managing_editor:
|
|
1052
1054
|
feed["author"] = managing_editor
|
|
1053
1055
|
|
|
1056
|
+
# Parse feed-level image/icon/logo
|
|
1057
|
+
if feed_type == "atom":
|
|
1058
|
+
icon_el = channel.find(f"{{{atom_ns}}}icon")
|
|
1059
|
+
if icon_el is not None and icon_el.text:
|
|
1060
|
+
feed["icon"] = icon_el.text.strip()
|
|
1061
|
+
logo_el = channel.find(f"{{{atom_ns}}}logo")
|
|
1062
|
+
if logo_el is not None and logo_el.text:
|
|
1063
|
+
feed["logo"] = logo_el.text.strip()
|
|
1064
|
+
elif feed_type == "rss":
|
|
1065
|
+
image_el = channel.find("image")
|
|
1066
|
+
if image_el is not None:
|
|
1067
|
+
image: dict[str, Optional[str]] = {}
|
|
1068
|
+
for sub_tag in ("url", "title", "link"):
|
|
1069
|
+
sub_el = image_el.find(sub_tag)
|
|
1070
|
+
if sub_el is not None and sub_el.text:
|
|
1071
|
+
image[sub_tag] = sub_el.text.strip()
|
|
1072
|
+
if image.get("url"):
|
|
1073
|
+
feed["image"] = image
|
|
1074
|
+
elif feed_type == "rdf":
|
|
1075
|
+
rdf_image_el = channel.find("{http://purl.org/rss/1.0/}image")
|
|
1076
|
+
if rdf_image_el is not None:
|
|
1077
|
+
image = {}
|
|
1078
|
+
for sub_tag in ("title", "link", "url"):
|
|
1079
|
+
sub_el = rdf_image_el.find(f"{{http://purl.org/rss/1.0/}}{sub_tag}")
|
|
1080
|
+
if sub_el is not None and sub_el.text:
|
|
1081
|
+
image[sub_tag] = sub_el.text.strip()
|
|
1082
|
+
if image.get("url"):
|
|
1083
|
+
feed["image"] = image
|
|
1084
|
+
|
|
1054
1085
|
# Parse feed-level tags/categories
|
|
1055
1086
|
if include_tags:
|
|
1056
1087
|
tags = _parse_tags(channel, feed_type, atom_ns)
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
"""Tests for feed-level image/icon/logo extraction."""
|
|
2
|
+
|
|
3
|
+
from fastfeedparser import parse
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def test_rss_feed_image():
|
|
7
|
+
xml = b"""<?xml version="1.0" encoding="UTF-8"?>
|
|
8
|
+
<rss version="2.0">
|
|
9
|
+
<channel>
|
|
10
|
+
<title>Test Feed</title>
|
|
11
|
+
<link>https://example.com</link>
|
|
12
|
+
<image>
|
|
13
|
+
<url>https://example.com/logo.png</url>
|
|
14
|
+
<title>Test Feed</title>
|
|
15
|
+
<link>https://example.com</link>
|
|
16
|
+
</image>
|
|
17
|
+
<item>
|
|
18
|
+
<title>Entry 1</title>
|
|
19
|
+
</item>
|
|
20
|
+
</channel>
|
|
21
|
+
</rss>"""
|
|
22
|
+
result = parse(xml)
|
|
23
|
+
assert result["feed"]["image"] == {
|
|
24
|
+
"url": "https://example.com/logo.png",
|
|
25
|
+
"title": "Test Feed",
|
|
26
|
+
"link": "https://example.com",
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def test_rss_feed_image_url_only():
|
|
31
|
+
xml = b"""<?xml version="1.0" encoding="UTF-8"?>
|
|
32
|
+
<rss version="2.0">
|
|
33
|
+
<channel>
|
|
34
|
+
<title>Test</title>
|
|
35
|
+
<image>
|
|
36
|
+
<url>https://example.com/icon.png</url>
|
|
37
|
+
</image>
|
|
38
|
+
<item><title>Entry</title></item>
|
|
39
|
+
</channel>
|
|
40
|
+
</rss>"""
|
|
41
|
+
result = parse(xml)
|
|
42
|
+
assert result["feed"]["image"]["url"] == "https://example.com/icon.png"
|
|
43
|
+
assert "title" not in result["feed"]["image"]
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def test_rss_feed_no_image():
|
|
47
|
+
xml = b"""<?xml version="1.0" encoding="UTF-8"?>
|
|
48
|
+
<rss version="2.0">
|
|
49
|
+
<channel>
|
|
50
|
+
<title>Test</title>
|
|
51
|
+
<item><title>Entry</title></item>
|
|
52
|
+
</channel>
|
|
53
|
+
</rss>"""
|
|
54
|
+
result = parse(xml)
|
|
55
|
+
assert "image" not in result["feed"]
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def test_rss_feed_image_empty_url():
|
|
59
|
+
xml = b"""<?xml version="1.0" encoding="UTF-8"?>
|
|
60
|
+
<rss version="2.0">
|
|
61
|
+
<channel>
|
|
62
|
+
<title>Test</title>
|
|
63
|
+
<image>
|
|
64
|
+
<url></url>
|
|
65
|
+
<title>Test</title>
|
|
66
|
+
</image>
|
|
67
|
+
<item><title>Entry</title></item>
|
|
68
|
+
</channel>
|
|
69
|
+
</rss>"""
|
|
70
|
+
result = parse(xml)
|
|
71
|
+
assert "image" not in result["feed"]
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def test_atom_feed_icon_and_logo():
|
|
75
|
+
xml = b"""<?xml version="1.0" encoding="UTF-8"?>
|
|
76
|
+
<feed xmlns="http://www.w3.org/2005/Atom">
|
|
77
|
+
<title>Test Feed</title>
|
|
78
|
+
<icon>https://example.com/favicon.ico</icon>
|
|
79
|
+
<logo>https://example.com/banner.png</logo>
|
|
80
|
+
<entry>
|
|
81
|
+
<title>Entry 1</title>
|
|
82
|
+
<id>urn:entry:1</id>
|
|
83
|
+
</entry>
|
|
84
|
+
</feed>"""
|
|
85
|
+
result = parse(xml)
|
|
86
|
+
assert result["feed"]["icon"] == "https://example.com/favicon.ico"
|
|
87
|
+
assert result["feed"]["logo"] == "https://example.com/banner.png"
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def test_atom_feed_icon_only():
|
|
91
|
+
xml = b"""<?xml version="1.0" encoding="UTF-8"?>
|
|
92
|
+
<feed xmlns="http://www.w3.org/2005/Atom">
|
|
93
|
+
<title>Test</title>
|
|
94
|
+
<icon>https://example.com/favicon.ico</icon>
|
|
95
|
+
<entry><title>E</title><id>urn:1</id></entry>
|
|
96
|
+
</feed>"""
|
|
97
|
+
result = parse(xml)
|
|
98
|
+
assert result["feed"]["icon"] == "https://example.com/favicon.ico"
|
|
99
|
+
assert "logo" not in result["feed"]
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def test_atom_feed_no_icon():
|
|
103
|
+
xml = b"""<?xml version="1.0" encoding="UTF-8"?>
|
|
104
|
+
<feed xmlns="http://www.w3.org/2005/Atom">
|
|
105
|
+
<title>Test</title>
|
|
106
|
+
<entry><title>E</title><id>urn:1</id></entry>
|
|
107
|
+
</feed>"""
|
|
108
|
+
result = parse(xml)
|
|
109
|
+
assert "icon" not in result["feed"]
|
|
110
|
+
assert "logo" not in result["feed"]
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def test_rdf_feed_image():
|
|
114
|
+
xml = b"""<?xml version="1.0" encoding="UTF-8"?>
|
|
115
|
+
<rdf:RDF
|
|
116
|
+
xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"
|
|
117
|
+
xmlns="http://purl.org/rss/1.0/">
|
|
118
|
+
<channel>
|
|
119
|
+
<title>Test RDF Feed</title>
|
|
120
|
+
<link>https://example.com</link>
|
|
121
|
+
</channel>
|
|
122
|
+
<image>
|
|
123
|
+
<url>https://example.com/rdf-logo.png</url>
|
|
124
|
+
<title>RDF Feed</title>
|
|
125
|
+
<link>https://example.com</link>
|
|
126
|
+
</image>
|
|
127
|
+
<item>
|
|
128
|
+
<title>Entry 1</title>
|
|
129
|
+
</item>
|
|
130
|
+
</rdf:RDF>"""
|
|
131
|
+
result = parse(xml)
|
|
132
|
+
assert result["feed"]["image"] == {
|
|
133
|
+
"url": "https://example.com/rdf-logo.png",
|
|
134
|
+
"title": "RDF Feed",
|
|
135
|
+
"link": "https://example.com",
|
|
136
|
+
}
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{fastfeedparser-0.5.9 → fastfeedparser-0.6.0}/src/fastfeedparser.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|