fastfeedparser 0.4.9__tar.gz → 0.5.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fastfeedparser-0.4.9/src/fastfeedparser.egg-info → fastfeedparser-0.5.1}/PKG-INFO +1 -1
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.1}/pyproject.toml +3 -0
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.1}/setup.cfg +1 -1
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.1}/src/fastfeedparser/__init__.py +1 -1
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.1}/src/fastfeedparser/main.py +61 -223
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.1/src/fastfeedparser.egg-info}/PKG-INFO +1 -1
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.1}/LICENSE +0 -0
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.1}/README.md +0 -0
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.1}/src/fastfeedparser.egg-info/SOURCES.txt +0 -0
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.1}/src/fastfeedparser.egg-info/dependency_links.txt +0 -0
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.1}/src/fastfeedparser.egg-info/requires.txt +0 -0
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.1}/src/fastfeedparser.egg-info/top_level.txt +0 -0
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.1}/tests/test_encoding.py +0 -0
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.1}/tests/test_integration.py +0 -0
|
@@ -3,6 +3,7 @@ from __future__ import annotations
|
|
|
3
3
|
import datetime
|
|
4
4
|
from email.utils import parsedate_to_datetime
|
|
5
5
|
import gzip
|
|
6
|
+
import html as _html_mod
|
|
6
7
|
import json
|
|
7
8
|
import re
|
|
8
9
|
import zlib
|
|
@@ -40,27 +41,18 @@ _RE_XML_DECL_ENCODING = re.compile(
|
|
|
40
41
|
_RE_XML_DECL_ENCODING_BYTES = re.compile(
|
|
41
42
|
br'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
|
|
42
43
|
)
|
|
43
|
-
_RE_DOUBLE_XML_DECL = re.compile(r"<\?xml\?xml\s+", re.IGNORECASE)
|
|
44
44
|
_RE_DOUBLE_XML_DECL_BYTES = re.compile(br"<\?xml\?xml\s+", re.IGNORECASE)
|
|
45
|
-
_RE_DOUBLE_CLOSE = re.compile(r"\?\?>\s*")
|
|
46
45
|
_RE_DOUBLE_CLOSE_BYTES = re.compile(br"\?\?>\s*")
|
|
47
|
-
_RE_UNQUOTED_ATTR = re.compile(r'(\s+[\w:]+)=([^\s>"\']+)')
|
|
48
46
|
_RE_UNQUOTED_ATTR_BYTES = re.compile(br'(\s+[\w:]+)=([^\s>"\']+)')
|
|
49
|
-
_RE_UTF16_ENCODING = re.compile(
|
|
50
|
-
r'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
|
|
51
|
-
)
|
|
52
47
|
_RE_UTF16_ENCODING_BYTES = re.compile(
|
|
53
48
|
br'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
|
|
54
49
|
)
|
|
55
|
-
_RE_UNCLOSED_LINK = re.compile(
|
|
56
|
-
r"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
|
|
57
|
-
)
|
|
58
50
|
_RE_UNCLOSED_LINK_BYTES = re.compile(
|
|
59
51
|
br"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
|
|
60
52
|
)
|
|
61
53
|
_RE_FEB29 = re.compile(r"(\d{4})-02-29")
|
|
54
|
+
_RE_HTML_TAGS = re.compile(r"<[^>]+>")
|
|
62
55
|
_RE_WHITESPACE = re.compile(r"\s+")
|
|
63
|
-
_RE_ISO_LIKE = re.compile(r"^\d{4}-\d{2}-\d{2}")
|
|
64
56
|
_RE_ISO_TZ_NO_COLON = re.compile(r"([+-]\d{2})(\d{2})$")
|
|
65
57
|
_RE_ISO_TZ_HOUR_ONLY = re.compile(r"([+-]\d{2})$")
|
|
66
58
|
_RE_ISO_FRACTION = re.compile(r"\.(\d{7,})(?=(?:[+-]\d{2}:?\d{2}|Z|$))", re.IGNORECASE)
|
|
@@ -111,36 +103,6 @@ def _ensure_utf8_xml_declaration(content: str) -> str:
|
|
|
111
103
|
return _RE_XML_DECL_ENCODING.sub(r"\1utf-8\3", content, count=1)
|
|
112
104
|
|
|
113
105
|
|
|
114
|
-
def _clean_feed_text(content: str) -> str:
|
|
115
|
-
"""Clean feed text by extracting the XML document (if it's embedded in junk)."""
|
|
116
|
-
stripped_content = content.lstrip()
|
|
117
|
-
stripped_lower = stripped_content[:2000].lower()
|
|
118
|
-
if stripped_lower.startswith(("<?xml", "<rss", "<feed", "<rdf")):
|
|
119
|
-
return stripped_content
|
|
120
|
-
|
|
121
|
-
if stripped_lower.startswith("<!doctype html") or stripped_lower.startswith("<html"):
|
|
122
|
-
raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
|
|
123
|
-
|
|
124
|
-
xml_start_patterns = (
|
|
125
|
-
"<?xml",
|
|
126
|
-
"<rss",
|
|
127
|
-
"<feed",
|
|
128
|
-
"<rdf:rdf",
|
|
129
|
-
"<?xml-stylesheet",
|
|
130
|
-
)
|
|
131
|
-
|
|
132
|
-
content_lines = content.splitlines()
|
|
133
|
-
for i, line in enumerate(content_lines):
|
|
134
|
-
line_stripped = line.strip().lower()
|
|
135
|
-
if any(line_stripped.startswith(pattern) for pattern in xml_start_patterns):
|
|
136
|
-
return "\n".join(content_lines[i:])
|
|
137
|
-
|
|
138
|
-
if "<script>" in stripped_lower or "<body>" in stripped_lower:
|
|
139
|
-
raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
|
|
140
|
-
|
|
141
|
-
return content
|
|
142
|
-
|
|
143
|
-
|
|
144
106
|
def _clean_feed_bytes(content: bytes) -> bytes:
|
|
145
107
|
"""Clean feed bytes by extracting the XML document (if it's embedded in junk)."""
|
|
146
108
|
stripped_content = content.lstrip()
|
|
@@ -223,60 +185,9 @@ def _prepare_xml_bytes(xml_content: str | bytes) -> bytes:
|
|
|
223
185
|
cleaned = _fix_malformed_xml_bytes(cleaned, actual_encoding=actual_encoding)
|
|
224
186
|
return cleaned
|
|
225
187
|
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
needs_fixing = (
|
|
231
|
-
"?xml?xml" in cleaned_text[:200]
|
|
232
|
-
or "??>" in cleaned_text[:200]
|
|
233
|
-
or (
|
|
234
|
-
"rss:" in cleaned_text[:500] and "xmlns:rss" not in cleaned_text[:1000]
|
|
235
|
-
)
|
|
236
|
-
or ("utf-16" in cleaned_text[:200].lower())
|
|
237
|
-
)
|
|
238
|
-
if needs_fixing:
|
|
239
|
-
cleaned_text = _fix_malformed_xml(cleaned_text, actual_encoding="utf-8")
|
|
240
|
-
|
|
241
|
-
cleaned_text = _ensure_utf8_xml_declaration(cleaned_text)
|
|
242
|
-
return cleaned_text.encode("utf-8", errors="replace")
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
def _fix_malformed_xml(content: str, actual_encoding: str = "utf-8") -> str:
|
|
246
|
-
"""Fix common malformed XML issues in feeds.
|
|
247
|
-
|
|
248
|
-
Some feeds have malformed XML like unclosed link tags or other issues
|
|
249
|
-
that can be automatically corrected.
|
|
250
|
-
|
|
251
|
-
Args:
|
|
252
|
-
content: The XML content as a string
|
|
253
|
-
actual_encoding: The actual encoding used (default: utf-8)
|
|
254
|
-
"""
|
|
255
|
-
# Fix double XML declarations like "<?xml?xml version="1.0"?>"
|
|
256
|
-
# This is found in dylanharris.org feed
|
|
257
|
-
content = _RE_DOUBLE_XML_DECL.sub(r"<?xml ", content)
|
|
258
|
-
|
|
259
|
-
# Fix double closing ?> in XML declaration like "??>>"
|
|
260
|
-
content = _RE_DOUBLE_CLOSE.sub(r"?>", content)
|
|
261
|
-
|
|
262
|
-
# Fix malformed attribute syntax like rss:version=2.0 (missing quotes)
|
|
263
|
-
# This is found in dylanharris.org feed
|
|
264
|
-
content = _RE_UNQUOTED_ATTR.sub(r'\1="\2"', content)
|
|
265
|
-
|
|
266
|
-
# Update encoding in XML declaration to match actual encoding
|
|
267
|
-
# This handles cases where content was transcoded from UTF-16 to UTF-8
|
|
268
|
-
if actual_encoding.lower() != "utf-16":
|
|
269
|
-
content = _RE_UTF16_ENCODING.sub(rf"\1{actual_encoding}\3", content)
|
|
270
|
-
|
|
271
|
-
# Fix unclosed link tags - common in Atom feeds
|
|
272
|
-
# Pattern: <link ...> followed by whitespace and another tag (not </link>)
|
|
273
|
-
# should be <link .../>
|
|
274
|
-
# Only fix link tags that are clearly malformed:
|
|
275
|
-
# - End with > instead of />
|
|
276
|
-
# - Are followed by whitespace and another tag (not a closing </link>)
|
|
277
|
-
content = _RE_UNCLOSED_LINK.sub(r"<link\1/>", content)
|
|
278
|
-
|
|
279
|
-
return content
|
|
188
|
+
# Str input: fix encoding declaration, encode to bytes, then use bytes path.
|
|
189
|
+
xml_content = _ensure_utf8_xml_declaration(xml_content)
|
|
190
|
+
return _prepare_xml_bytes(xml_content.encode("utf-8", errors="replace"))
|
|
280
191
|
|
|
281
192
|
|
|
282
193
|
def _parse_json_feed(json_data: dict) -> FastFeedParserDict:
|
|
@@ -555,47 +466,31 @@ def _extract_error_message(root: _Element, raw_bytes: Optional[bytes] = None) ->
|
|
|
555
466
|
return error_msg
|
|
556
467
|
|
|
557
468
|
|
|
469
|
+
_NON_FEED_MESSAGES: dict[str, str] = {
|
|
470
|
+
"html": "Received HTML page instead of feed",
|
|
471
|
+
"div": "Received HTML fragment instead of feed",
|
|
472
|
+
"body": "Received HTML fragment instead of feed",
|
|
473
|
+
"br": "Received HTML fragment instead of feed",
|
|
474
|
+
"status": "Feed server returned status message",
|
|
475
|
+
"error": "Feed server returned error",
|
|
476
|
+
"opml": "Received OPML document instead of feed (OPML is an outline format, not a feed)",
|
|
477
|
+
"urlset": "Received XML sitemap instead of feed (sitemap is for search engines, not a feed)",
|
|
478
|
+
"sitemapindex": "Received XML sitemap instead of feed (sitemap is for search engines, not a feed)",
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
|
|
558
482
|
def _raise_for_non_feed_root(
|
|
559
483
|
root: _Element, root_tag_local: str, raw_bytes: Optional[bytes] = None
|
|
560
484
|
) -> None:
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
"urlset", "sitemapindex",
|
|
564
|
-
}
|
|
565
|
-
if root_tag_local not in non_feed_tags:
|
|
485
|
+
base_msg = _NON_FEED_MESSAGES.get(root_tag_local)
|
|
486
|
+
if base_msg is None:
|
|
566
487
|
return
|
|
567
488
|
|
|
568
489
|
error_msg = _extract_error_message(root, raw_bytes).strip()[:300] or "No error message"
|
|
569
490
|
|
|
570
|
-
if
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
raise ValueError(
|
|
574
|
-
"Received HTML page instead of feed (possible redirect, 404, or server error)"
|
|
575
|
-
)
|
|
576
|
-
if root_tag_local in {"div", "body"}:
|
|
577
|
-
if error_msg != "No error message" and len(error_msg) > 10:
|
|
578
|
-
raise ValueError(f"Received HTML fragment instead of feed: {error_msg[:150]}")
|
|
579
|
-
raise ValueError("Received HTML fragment instead of feed")
|
|
580
|
-
if root_tag_local == "br":
|
|
581
|
-
if error_msg != "No error message" and len(error_msg) > 10:
|
|
582
|
-
raise ValueError(f"Received HTML error instead of feed: {error_msg[:150]}")
|
|
583
|
-
raise ValueError("Received HTML fragment instead of feed")
|
|
584
|
-
if root_tag_local == "status":
|
|
585
|
-
raise ValueError(f"Feed server returned status message: {error_msg}")
|
|
586
|
-
if root_tag_local == "error":
|
|
587
|
-
if error_msg != "No error message":
|
|
588
|
-
raise ValueError(f"Feed server returned error: {error_msg}")
|
|
589
|
-
raise ValueError("Feed server returned error (no details provided)")
|
|
590
|
-
if root_tag_local == "opml":
|
|
591
|
-
raise ValueError(
|
|
592
|
-
"Received OPML document instead of feed (OPML is an outline format, not a feed)"
|
|
593
|
-
)
|
|
594
|
-
if root_tag_local in {"urlset", "sitemapindex"}:
|
|
595
|
-
raise ValueError(
|
|
596
|
-
"Received XML sitemap instead of feed (sitemap is for search engines, not a feed)"
|
|
597
|
-
)
|
|
598
|
-
raise ValueError(f"Not a valid feed: {root_tag_local} element found - {error_msg[:100]}")
|
|
491
|
+
if error_msg != "No error message" and len(error_msg) > 10:
|
|
492
|
+
raise ValueError(f"{base_msg}: {error_msg[:150]}")
|
|
493
|
+
raise ValueError(base_msg)
|
|
599
494
|
|
|
600
495
|
|
|
601
496
|
_RE_META_REFRESH_URL = re.compile(
|
|
@@ -730,24 +625,6 @@ def _detect_feed_structure(
|
|
|
730
625
|
raise ValueError(f"Unknown feed type: {root.tag}")
|
|
731
626
|
|
|
732
627
|
|
|
733
|
-
def _should_parse_media_content(root: _Element, xml_content: bytes) -> bool:
|
|
734
|
-
"""Check if feed likely contains Media RSS fields."""
|
|
735
|
-
ns_values = root.nsmap.values() if root.nsmap else ()
|
|
736
|
-
for ns_value in ns_values:
|
|
737
|
-
if not ns_value:
|
|
738
|
-
continue
|
|
739
|
-
if "search.yahoo.com/mrss" in ns_value:
|
|
740
|
-
return True
|
|
741
|
-
|
|
742
|
-
# Fallback for feeds with undeclared/late namespace usage.
|
|
743
|
-
return b"search.yahoo.com/mrss" in xml_content or b"<media:" in xml_content
|
|
744
|
-
|
|
745
|
-
|
|
746
|
-
def _should_parse_enclosures(feed_type: _FeedType, xml_content: bytes) -> bool:
|
|
747
|
-
"""Check if feed likely contains RSS enclosure elements."""
|
|
748
|
-
return feed_type == "rss" and b"<enclosure" in xml_content
|
|
749
|
-
|
|
750
|
-
|
|
751
628
|
def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
752
629
|
"""Parse feed content (XML or JSON) that has already been fetched."""
|
|
753
630
|
json_feed = _maybe_parse_json_feed(xml_content)
|
|
@@ -762,8 +639,6 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
|
762
639
|
feed_type, channel, items, atom_namespace = _detect_feed_structure(
|
|
763
640
|
root, xml_content, root_tag_local
|
|
764
641
|
)
|
|
765
|
-
parse_media_content = _should_parse_media_content(root, xml_content)
|
|
766
|
-
parse_enclosures = _should_parse_enclosures(feed_type, xml_content)
|
|
767
642
|
|
|
768
643
|
feed = _parse_feed_info(channel, feed_type, atom_namespace)
|
|
769
644
|
|
|
@@ -775,8 +650,6 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
|
775
650
|
item,
|
|
776
651
|
feed_type,
|
|
777
652
|
atom_namespace,
|
|
778
|
-
parse_media_content=parse_media_content,
|
|
779
|
-
parse_enclosures=parse_enclosures,
|
|
780
653
|
)
|
|
781
654
|
# Ensure that titles and descriptions are always present
|
|
782
655
|
entry["title"] = entry.get("title", "").strip()
|
|
@@ -1123,16 +996,9 @@ def _populate_entry_content(
|
|
|
1123
996
|
content_value = entry["content"][0]["value"]
|
|
1124
997
|
if content_value:
|
|
1125
998
|
if "<" in content_value:
|
|
1126
|
-
|
|
1127
|
-
|
|
1128
|
-
|
|
1129
|
-
content_text = html_content.xpath("string()")
|
|
1130
|
-
if isinstance(content_text, str):
|
|
1131
|
-
content_value = _RE_WHITESPACE.sub(" ", content_text)
|
|
1132
|
-
except etree.ParserError:
|
|
1133
|
-
pass
|
|
1134
|
-
else:
|
|
1135
|
-
content_value = _RE_WHITESPACE.sub(" ", content_value)
|
|
999
|
+
content_value = _RE_HTML_TAGS.sub(" ", content_value[:2048])
|
|
1000
|
+
content_value = _html_mod.unescape(content_value)
|
|
1001
|
+
content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
|
|
1136
1002
|
entry["description"] = content_value[:512]
|
|
1137
1003
|
|
|
1138
1004
|
|
|
@@ -1223,13 +1089,6 @@ def _parse_enclosures(item: _Element) -> list[dict[str, Any]] | None:
|
|
|
1223
1089
|
return enclosures or None
|
|
1224
1090
|
|
|
1225
1091
|
|
|
1226
|
-
def _normalize_local_tag_name(tag: str) -> str:
|
|
1227
|
-
local = tag.rsplit("}", 1)[-1].lower()
|
|
1228
|
-
if ":" in local:
|
|
1229
|
-
local = local.split(":", 1)[1]
|
|
1230
|
-
return local
|
|
1231
|
-
|
|
1232
|
-
|
|
1233
1092
|
def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]], dict[str, Optional[str]]]:
|
|
1234
1093
|
by_local: dict[str, Optional[str]] = {}
|
|
1235
1094
|
by_full: dict[str, Optional[str]] = {}
|
|
@@ -1240,7 +1099,9 @@ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]],
|
|
|
1240
1099
|
text_value = child.text.strip() if child.text else None
|
|
1241
1100
|
if tag not in by_full:
|
|
1242
1101
|
by_full[tag] = text_value
|
|
1243
|
-
local =
|
|
1102
|
+
local = tag.rsplit("}", 1)[-1].lower()
|
|
1103
|
+
if ":" in local:
|
|
1104
|
+
local = local.split(":", 1)[1]
|
|
1244
1105
|
if local not in by_local:
|
|
1245
1106
|
by_local[local] = text_value
|
|
1246
1107
|
return by_local, by_full
|
|
@@ -1257,8 +1118,6 @@ def _first_non_empty(mapping: dict[str, Optional[str]], keys: tuple[str, ...]) -
|
|
|
1257
1118
|
def _parse_rss_feed_entry_fast(
|
|
1258
1119
|
item: _Element,
|
|
1259
1120
|
atom_ns: str,
|
|
1260
|
-
parse_media_content: bool = True,
|
|
1261
|
-
parse_enclosures: bool = True,
|
|
1262
1121
|
) -> FastFeedParserDict:
|
|
1263
1122
|
text_by_local, text_by_full = _build_rss_item_text_maps(item)
|
|
1264
1123
|
|
|
@@ -1294,7 +1153,7 @@ def _parse_rss_feed_entry_fast(
|
|
|
1294
1153
|
if updated:
|
|
1295
1154
|
entry["updated"] = updated
|
|
1296
1155
|
|
|
1297
|
-
if "published" not in entry and rss_guid:
|
|
1156
|
+
if "published" not in entry and rss_guid and not rss_guid.startswith(("http://", "https://")):
|
|
1298
1157
|
guid_date = _parse_date(rss_guid)
|
|
1299
1158
|
if guid_date:
|
|
1300
1159
|
entry["published"] = guid_date
|
|
@@ -1308,15 +1167,13 @@ def _parse_rss_feed_entry_fast(
|
|
|
1308
1167
|
|
|
1309
1168
|
_populate_entry_content(entry, item, "rss", atom_ns)
|
|
1310
1169
|
|
|
1311
|
-
|
|
1312
|
-
|
|
1313
|
-
|
|
1314
|
-
entry["media_content"] = media_contents
|
|
1170
|
+
media_contents = _parse_media_content(item)
|
|
1171
|
+
if media_contents:
|
|
1172
|
+
entry["media_content"] = media_contents
|
|
1315
1173
|
|
|
1316
|
-
|
|
1317
|
-
|
|
1318
|
-
|
|
1319
|
-
entry["enclosures"] = enclosures
|
|
1174
|
+
enclosures = _parse_enclosures(item)
|
|
1175
|
+
if enclosures:
|
|
1176
|
+
entry["enclosures"] = enclosures
|
|
1320
1177
|
|
|
1321
1178
|
author = _first_non_empty(text_by_local, ("author", "creator"))
|
|
1322
1179
|
if not author:
|
|
@@ -1340,20 +1197,12 @@ def _parse_feed_entry(
|
|
|
1340
1197
|
item: _Element,
|
|
1341
1198
|
feed_type: _FeedType,
|
|
1342
1199
|
atom_namespace: Optional[str] = None,
|
|
1343
|
-
*,
|
|
1344
|
-
parse_media_content: bool = True,
|
|
1345
|
-
parse_enclosures: bool = True,
|
|
1346
1200
|
) -> FastFeedParserDict:
|
|
1347
1201
|
# Use dynamic atom namespace or fallback to default
|
|
1348
1202
|
atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
|
|
1349
1203
|
|
|
1350
1204
|
if feed_type == "rss":
|
|
1351
|
-
return _parse_rss_feed_entry_fast(
|
|
1352
|
-
item,
|
|
1353
|
-
atom_ns,
|
|
1354
|
-
parse_media_content=parse_media_content,
|
|
1355
|
-
parse_enclosures=parse_enclosures,
|
|
1356
|
-
)
|
|
1205
|
+
return _parse_rss_feed_entry_fast(item, atom_ns)
|
|
1357
1206
|
|
|
1358
1207
|
# Check if this is Atom 0.3 to use different date field names
|
|
1359
1208
|
is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
|
|
@@ -1446,8 +1295,7 @@ def _parse_feed_entry(
|
|
|
1446
1295
|
entry["updated"] = _parse_date(fallback_updated)
|
|
1447
1296
|
|
|
1448
1297
|
# Try to extract date from GUID as final fallback
|
|
1449
|
-
if "published" not in entry and rss_guid:
|
|
1450
|
-
# Check if GUID contains date information
|
|
1298
|
+
if "published" not in entry and rss_guid and not rss_guid.startswith(("http://", "https://")):
|
|
1451
1299
|
guid_date = _parse_date(rss_guid)
|
|
1452
1300
|
if guid_date:
|
|
1453
1301
|
entry["published"] = guid_date
|
|
@@ -1467,32 +1315,22 @@ def _parse_feed_entry(
|
|
|
1467
1315
|
|
|
1468
1316
|
_populate_entry_content(entry, item, feed_type, atom_ns)
|
|
1469
1317
|
|
|
1470
|
-
|
|
1471
|
-
|
|
1472
|
-
|
|
1473
|
-
entry["media_content"] = media_contents
|
|
1318
|
+
media_contents = _parse_media_content(item)
|
|
1319
|
+
if media_contents:
|
|
1320
|
+
entry["media_content"] = media_contents
|
|
1474
1321
|
|
|
1475
|
-
|
|
1476
|
-
|
|
1477
|
-
|
|
1478
|
-
entry["enclosures"] = enclosures
|
|
1322
|
+
enclosures = _parse_enclosures(item)
|
|
1323
|
+
if enclosures:
|
|
1324
|
+
entry["enclosures"] = enclosures
|
|
1479
1325
|
|
|
1480
|
-
author = (
|
|
1481
|
-
|
|
1482
|
-
|
|
1483
|
-
|
|
1484
|
-
|
|
1485
|
-
False,
|
|
1486
|
-
)
|
|
1487
|
-
or get_field_value(
|
|
1488
|
-
"{http://purl.org/dc/elements/1.1/}creator",
|
|
1489
|
-
"{http://purl.org/dc/elements/1.1/}creator",
|
|
1490
|
-
"{http://purl.org/dc/elements/1.1/}creator",
|
|
1491
|
-
False,
|
|
1492
|
-
)
|
|
1493
|
-
or element_get("{http://purl.org/dc/elements/1.1/}creator")
|
|
1494
|
-
or element_get("author")
|
|
1326
|
+
author = get_field_value(
|
|
1327
|
+
"author",
|
|
1328
|
+
f"{{{atom_ns}}}author/{{{atom_ns}}}name",
|
|
1329
|
+
"{http://purl.org/dc/elements/1.1/}creator",
|
|
1330
|
+
False,
|
|
1495
1331
|
)
|
|
1332
|
+
if not author:
|
|
1333
|
+
author = element_get("{http://purl.org/dc/elements/1.1/}creator") or element_get("author")
|
|
1496
1334
|
if author:
|
|
1497
1335
|
entry["author"] = author
|
|
1498
1336
|
|
|
@@ -1648,7 +1486,7 @@ def _normalize_iso_datetime_string(value: str) -> str:
|
|
|
1648
1486
|
if cleaned.endswith(("Z", "z")):
|
|
1649
1487
|
cleaned = cleaned[:-1] + "+00:00"
|
|
1650
1488
|
|
|
1651
|
-
if " " in cleaned and "T" not in cleaned[:11] and
|
|
1489
|
+
if " " in cleaned and "T" not in cleaned[:11] and len(cleaned) >= 10 and cleaned[4] == "-" and cleaned[0:4].isdigit():
|
|
1652
1490
|
date_part, rest = cleaned.split(" ", 1)
|
|
1653
1491
|
if rest and rest[0].isdigit():
|
|
1654
1492
|
cleaned = f"{date_part}T{rest}"
|
|
@@ -1684,7 +1522,7 @@ def _parsedate_to_utc(value: str) -> Optional[datetime.datetime]:
|
|
|
1684
1522
|
return _ensure_utc(parsed)
|
|
1685
1523
|
|
|
1686
1524
|
|
|
1687
|
-
|
|
1525
|
+
_custom_tzinfos: dict[str, int] = {
|
|
1688
1526
|
"UTC": 0,
|
|
1689
1527
|
"UT": 0,
|
|
1690
1528
|
"GMT": 0,
|
|
@@ -1735,7 +1573,7 @@ _DATEPARSER_SETTINGS = {
|
|
|
1735
1573
|
@lru_cache(maxsize=512)
|
|
1736
1574
|
def _slow_dateutil_parse(value: str) -> Optional[datetime.datetime]:
|
|
1737
1575
|
try:
|
|
1738
|
-
return dateutil_parser.parse(value, tzinfos=
|
|
1576
|
+
return dateutil_parser.parse(value, tzinfos=_custom_tzinfos, ignoretz=False)
|
|
1739
1577
|
except (ValueError, TypeError, OverflowError):
|
|
1740
1578
|
return None
|
|
1741
1579
|
|
|
@@ -1772,12 +1610,12 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1772
1610
|
|
|
1773
1611
|
# Fix invalid leap year dates (Feb 29 in non-leap years)
|
|
1774
1612
|
# This handles feeds with incorrect dates like "2023-02-29"
|
|
1775
|
-
|
|
1776
|
-
|
|
1777
|
-
|
|
1778
|
-
|
|
1779
|
-
|
|
1780
|
-
|
|
1613
|
+
if "-02-29" in candidate:
|
|
1614
|
+
year_match = _RE_FEB29.match(candidate)
|
|
1615
|
+
if year_match:
|
|
1616
|
+
year = int(year_match.group(1))
|
|
1617
|
+
if not ((year % 4 == 0 and year % 100 != 0) or (year % 400 == 0)):
|
|
1618
|
+
candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
|
|
1781
1619
|
|
|
1782
1620
|
if "24:00" in candidate:
|
|
1783
1621
|
candidate = candidate.replace("24:00:00", "00:00:00").replace(
|
|
@@ -1786,7 +1624,7 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1786
1624
|
|
|
1787
1625
|
dt: Optional[datetime.datetime] = None
|
|
1788
1626
|
|
|
1789
|
-
is_iso_like =
|
|
1627
|
+
is_iso_like = len(candidate) >= 10 and candidate[4] == "-" and candidate[0:4].isdigit()
|
|
1790
1628
|
if is_iso_like:
|
|
1791
1629
|
iso_candidate = _normalize_iso_datetime_string(candidate)
|
|
1792
1630
|
try:
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{fastfeedparser-0.4.9 → fastfeedparser-0.5.1}/src/fastfeedparser.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|