fastfeedparser 0.5.0__tar.gz → 0.5.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fastfeedparser-0.5.0/src/fastfeedparser.egg-info → fastfeedparser-0.5.2}/PKG-INFO +1 -1
- {fastfeedparser-0.5.0 → fastfeedparser-0.5.2}/pyproject.toml +3 -0
- {fastfeedparser-0.5.0 → fastfeedparser-0.5.2}/setup.cfg +1 -1
- {fastfeedparser-0.5.0 → fastfeedparser-0.5.2}/src/fastfeedparser/__init__.py +1 -1
- {fastfeedparser-0.5.0 → fastfeedparser-0.5.2}/src/fastfeedparser/main.py +169 -185
- {fastfeedparser-0.5.0 → fastfeedparser-0.5.2/src/fastfeedparser.egg-info}/PKG-INFO +1 -1
- {fastfeedparser-0.5.0 → fastfeedparser-0.5.2}/LICENSE +0 -0
- {fastfeedparser-0.5.0 → fastfeedparser-0.5.2}/README.md +0 -0
- {fastfeedparser-0.5.0 → fastfeedparser-0.5.2}/src/fastfeedparser.egg-info/SOURCES.txt +0 -0
- {fastfeedparser-0.5.0 → fastfeedparser-0.5.2}/src/fastfeedparser.egg-info/dependency_links.txt +0 -0
- {fastfeedparser-0.5.0 → fastfeedparser-0.5.2}/src/fastfeedparser.egg-info/requires.txt +0 -0
- {fastfeedparser-0.5.0 → fastfeedparser-0.5.2}/src/fastfeedparser.egg-info/top_level.txt +0 -0
- {fastfeedparser-0.5.0 → fastfeedparser-0.5.2}/tests/test_encoding.py +0 -0
- {fastfeedparser-0.5.0 → fastfeedparser-0.5.2}/tests/test_integration.py +0 -0
|
@@ -41,21 +41,12 @@ _RE_XML_DECL_ENCODING = re.compile(
|
|
|
41
41
|
_RE_XML_DECL_ENCODING_BYTES = re.compile(
|
|
42
42
|
br'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
|
|
43
43
|
)
|
|
44
|
-
_RE_DOUBLE_XML_DECL = re.compile(r"<\?xml\?xml\s+", re.IGNORECASE)
|
|
45
44
|
_RE_DOUBLE_XML_DECL_BYTES = re.compile(br"<\?xml\?xml\s+", re.IGNORECASE)
|
|
46
|
-
_RE_DOUBLE_CLOSE = re.compile(r"\?\?>\s*")
|
|
47
45
|
_RE_DOUBLE_CLOSE_BYTES = re.compile(br"\?\?>\s*")
|
|
48
|
-
_RE_UNQUOTED_ATTR = re.compile(r'(\s+[\w:]+)=([^\s>"\']+)')
|
|
49
46
|
_RE_UNQUOTED_ATTR_BYTES = re.compile(br'(\s+[\w:]+)=([^\s>"\']+)')
|
|
50
|
-
_RE_UTF16_ENCODING = re.compile(
|
|
51
|
-
r'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
|
|
52
|
-
)
|
|
53
47
|
_RE_UTF16_ENCODING_BYTES = re.compile(
|
|
54
48
|
br'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
|
|
55
49
|
)
|
|
56
|
-
_RE_UNCLOSED_LINK = re.compile(
|
|
57
|
-
r"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
|
|
58
|
-
)
|
|
59
50
|
_RE_UNCLOSED_LINK_BYTES = re.compile(
|
|
60
51
|
br"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
|
|
61
52
|
)
|
|
@@ -112,36 +103,6 @@ def _ensure_utf8_xml_declaration(content: str) -> str:
|
|
|
112
103
|
return _RE_XML_DECL_ENCODING.sub(r"\1utf-8\3", content, count=1)
|
|
113
104
|
|
|
114
105
|
|
|
115
|
-
def _clean_feed_text(content: str) -> str:
|
|
116
|
-
"""Clean feed text by extracting the XML document (if it's embedded in junk)."""
|
|
117
|
-
stripped_content = content.lstrip()
|
|
118
|
-
stripped_lower = stripped_content[:2000].lower()
|
|
119
|
-
if stripped_lower.startswith(("<?xml", "<rss", "<feed", "<rdf")):
|
|
120
|
-
return stripped_content
|
|
121
|
-
|
|
122
|
-
if stripped_lower.startswith("<!doctype html") or stripped_lower.startswith("<html"):
|
|
123
|
-
raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
|
|
124
|
-
|
|
125
|
-
xml_start_patterns = (
|
|
126
|
-
"<?xml",
|
|
127
|
-
"<rss",
|
|
128
|
-
"<feed",
|
|
129
|
-
"<rdf:rdf",
|
|
130
|
-
"<?xml-stylesheet",
|
|
131
|
-
)
|
|
132
|
-
|
|
133
|
-
content_lines = content.splitlines()
|
|
134
|
-
for i, line in enumerate(content_lines):
|
|
135
|
-
line_stripped = line.strip().lower()
|
|
136
|
-
if any(line_stripped.startswith(pattern) for pattern in xml_start_patterns):
|
|
137
|
-
return "\n".join(content_lines[i:])
|
|
138
|
-
|
|
139
|
-
if "<script>" in stripped_lower or "<body>" in stripped_lower:
|
|
140
|
-
raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
|
|
141
|
-
|
|
142
|
-
return content
|
|
143
|
-
|
|
144
|
-
|
|
145
106
|
def _clean_feed_bytes(content: bytes) -> bytes:
|
|
146
107
|
"""Clean feed bytes by extracting the XML document (if it's embedded in junk)."""
|
|
147
108
|
stripped_content = content.lstrip()
|
|
@@ -224,60 +185,9 @@ def _prepare_xml_bytes(xml_content: str | bytes) -> bytes:
|
|
|
224
185
|
cleaned = _fix_malformed_xml_bytes(cleaned, actual_encoding=actual_encoding)
|
|
225
186
|
return cleaned
|
|
226
187
|
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
needs_fixing = (
|
|
232
|
-
"?xml?xml" in cleaned_text[:200]
|
|
233
|
-
or "??>" in cleaned_text[:200]
|
|
234
|
-
or (
|
|
235
|
-
"rss:" in cleaned_text[:500] and "xmlns:rss" not in cleaned_text[:1000]
|
|
236
|
-
)
|
|
237
|
-
or ("utf-16" in cleaned_text[:200].lower())
|
|
238
|
-
)
|
|
239
|
-
if needs_fixing:
|
|
240
|
-
cleaned_text = _fix_malformed_xml(cleaned_text, actual_encoding="utf-8")
|
|
241
|
-
|
|
242
|
-
cleaned_text = _ensure_utf8_xml_declaration(cleaned_text)
|
|
243
|
-
return cleaned_text.encode("utf-8", errors="replace")
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
def _fix_malformed_xml(content: str, actual_encoding: str = "utf-8") -> str:
|
|
247
|
-
"""Fix common malformed XML issues in feeds.
|
|
248
|
-
|
|
249
|
-
Some feeds have malformed XML like unclosed link tags or other issues
|
|
250
|
-
that can be automatically corrected.
|
|
251
|
-
|
|
252
|
-
Args:
|
|
253
|
-
content: The XML content as a string
|
|
254
|
-
actual_encoding: The actual encoding used (default: utf-8)
|
|
255
|
-
"""
|
|
256
|
-
# Fix double XML declarations like "<?xml?xml version="1.0"?>"
|
|
257
|
-
# This is found in dylanharris.org feed
|
|
258
|
-
content = _RE_DOUBLE_XML_DECL.sub(r"<?xml ", content)
|
|
259
|
-
|
|
260
|
-
# Fix double closing ?> in XML declaration like "??>>"
|
|
261
|
-
content = _RE_DOUBLE_CLOSE.sub(r"?>", content)
|
|
262
|
-
|
|
263
|
-
# Fix malformed attribute syntax like rss:version=2.0 (missing quotes)
|
|
264
|
-
# This is found in dylanharris.org feed
|
|
265
|
-
content = _RE_UNQUOTED_ATTR.sub(r'\1="\2"', content)
|
|
266
|
-
|
|
267
|
-
# Update encoding in XML declaration to match actual encoding
|
|
268
|
-
# This handles cases where content was transcoded from UTF-16 to UTF-8
|
|
269
|
-
if actual_encoding.lower() != "utf-16":
|
|
270
|
-
content = _RE_UTF16_ENCODING.sub(rf"\1{actual_encoding}\3", content)
|
|
271
|
-
|
|
272
|
-
# Fix unclosed link tags - common in Atom feeds
|
|
273
|
-
# Pattern: <link ...> followed by whitespace and another tag (not </link>)
|
|
274
|
-
# should be <link .../>
|
|
275
|
-
# Only fix link tags that are clearly malformed:
|
|
276
|
-
# - End with > instead of />
|
|
277
|
-
# - Are followed by whitespace and another tag (not a closing </link>)
|
|
278
|
-
content = _RE_UNCLOSED_LINK.sub(r"<link\1/>", content)
|
|
279
|
-
|
|
280
|
-
return content
|
|
188
|
+
# Str input: fix encoding declaration, encode to bytes, then use bytes path.
|
|
189
|
+
xml_content = _ensure_utf8_xml_declaration(xml_content)
|
|
190
|
+
return _prepare_xml_bytes(xml_content.encode("utf-8", errors="replace"))
|
|
281
191
|
|
|
282
192
|
|
|
283
193
|
def _parse_json_feed(json_data: dict) -> FastFeedParserDict:
|
|
@@ -556,46 +466,31 @@ def _extract_error_message(root: _Element, raw_bytes: Optional[bytes] = None) ->
|
|
|
556
466
|
return error_msg
|
|
557
467
|
|
|
558
468
|
|
|
469
|
+
_NON_FEED_MESSAGES: dict[str, str] = {
|
|
470
|
+
"html": "Received HTML page instead of feed",
|
|
471
|
+
"div": "Received HTML fragment instead of feed",
|
|
472
|
+
"body": "Received HTML fragment instead of feed",
|
|
473
|
+
"br": "Received HTML fragment instead of feed",
|
|
474
|
+
"status": "Feed server returned status message",
|
|
475
|
+
"error": "Feed server returned error",
|
|
476
|
+
"opml": "Received OPML document instead of feed (OPML is an outline format, not a feed)",
|
|
477
|
+
"urlset": "Received XML sitemap instead of feed (sitemap is for search engines, not a feed)",
|
|
478
|
+
"sitemapindex": "Received XML sitemap instead of feed (sitemap is for search engines, not a feed)",
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
|
|
559
482
|
def _raise_for_non_feed_root(
|
|
560
483
|
root: _Element, root_tag_local: str, raw_bytes: Optional[bytes] = None
|
|
561
484
|
) -> None:
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
"urlset", "sitemapindex",
|
|
565
|
-
}
|
|
566
|
-
if root_tag_local not in non_feed_tags:
|
|
485
|
+
base_msg = _NON_FEED_MESSAGES.get(root_tag_local)
|
|
486
|
+
if base_msg is None:
|
|
567
487
|
return
|
|
568
488
|
|
|
569
489
|
error_msg = _extract_error_message(root, raw_bytes).strip()[:300] or "No error message"
|
|
570
490
|
|
|
571
|
-
if
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
raise ValueError(
|
|
575
|
-
"Received HTML page instead of feed (possible redirect, 404, or server error)"
|
|
576
|
-
)
|
|
577
|
-
if root_tag_local in {"div", "body"}:
|
|
578
|
-
if error_msg != "No error message" and len(error_msg) > 10:
|
|
579
|
-
raise ValueError(f"Received HTML fragment instead of feed: {error_msg[:150]}")
|
|
580
|
-
raise ValueError("Received HTML fragment instead of feed")
|
|
581
|
-
if root_tag_local == "br":
|
|
582
|
-
if error_msg != "No error message" and len(error_msg) > 10:
|
|
583
|
-
raise ValueError(f"Received HTML error instead of feed: {error_msg[:150]}")
|
|
584
|
-
raise ValueError("Received HTML fragment instead of feed")
|
|
585
|
-
if root_tag_local == "status":
|
|
586
|
-
raise ValueError(f"Feed server returned status message: {error_msg}")
|
|
587
|
-
if root_tag_local == "error":
|
|
588
|
-
if error_msg != "No error message":
|
|
589
|
-
raise ValueError(f"Feed server returned error: {error_msg}")
|
|
590
|
-
raise ValueError("Feed server returned error (no details provided)")
|
|
591
|
-
if root_tag_local == "opml":
|
|
592
|
-
raise ValueError(
|
|
593
|
-
"Received OPML document instead of feed (OPML is an outline format, not a feed)"
|
|
594
|
-
)
|
|
595
|
-
if root_tag_local in {"urlset", "sitemapindex"}:
|
|
596
|
-
raise ValueError(
|
|
597
|
-
"Received XML sitemap instead of feed (sitemap is for search engines, not a feed)"
|
|
598
|
-
)
|
|
491
|
+
if error_msg != "No error message" and len(error_msg) > 10:
|
|
492
|
+
raise ValueError(f"{base_msg}: {error_msg[:150]}")
|
|
493
|
+
raise ValueError(base_msg)
|
|
599
494
|
|
|
600
495
|
|
|
601
496
|
_RE_META_REFRESH_URL = re.compile(
|
|
@@ -730,24 +625,6 @@ def _detect_feed_structure(
|
|
|
730
625
|
raise ValueError(f"Unknown feed type: {root.tag}")
|
|
731
626
|
|
|
732
627
|
|
|
733
|
-
def _should_parse_media_content(root: _Element, xml_content: bytes) -> bool:
|
|
734
|
-
"""Check if feed likely contains Media RSS fields."""
|
|
735
|
-
ns_values = root.nsmap.values() if root.nsmap else ()
|
|
736
|
-
for ns_value in ns_values:
|
|
737
|
-
if not ns_value:
|
|
738
|
-
continue
|
|
739
|
-
if "search.yahoo.com/mrss" in ns_value:
|
|
740
|
-
return True
|
|
741
|
-
|
|
742
|
-
# Fallback for feeds with undeclared/late namespace usage.
|
|
743
|
-
return b"search.yahoo.com/mrss" in xml_content or b"<media:" in xml_content
|
|
744
|
-
|
|
745
|
-
|
|
746
|
-
def _should_parse_enclosures(feed_type: _FeedType, xml_content: bytes) -> bool:
|
|
747
|
-
"""Check if feed likely contains RSS enclosure elements."""
|
|
748
|
-
return feed_type == "rss" and b"<enclosure" in xml_content
|
|
749
|
-
|
|
750
|
-
|
|
751
628
|
def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
752
629
|
"""Parse feed content (XML or JSON) that has already been fetched."""
|
|
753
630
|
json_feed = _maybe_parse_json_feed(xml_content)
|
|
@@ -762,8 +639,6 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
|
762
639
|
feed_type, channel, items, atom_namespace = _detect_feed_structure(
|
|
763
640
|
root, xml_content, root_tag_local
|
|
764
641
|
)
|
|
765
|
-
parse_media_content = _should_parse_media_content(root, xml_content)
|
|
766
|
-
parse_enclosures = _should_parse_enclosures(feed_type, xml_content)
|
|
767
642
|
|
|
768
643
|
feed = _parse_feed_info(channel, feed_type, atom_namespace)
|
|
769
644
|
|
|
@@ -775,8 +650,6 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
|
775
650
|
item,
|
|
776
651
|
feed_type,
|
|
777
652
|
atom_namespace,
|
|
778
|
-
parse_media_content=parse_media_content,
|
|
779
|
-
parse_enclosures=parse_enclosures,
|
|
780
653
|
)
|
|
781
654
|
# Ensure that titles and descriptions are always present
|
|
782
655
|
entry["title"] = entry.get("title", "").strip()
|
|
@@ -1223,7 +1096,7 @@ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]],
|
|
|
1223
1096
|
tag = child.tag
|
|
1224
1097
|
if not isinstance(tag, str):
|
|
1225
1098
|
continue
|
|
1226
|
-
text_value = child.text
|
|
1099
|
+
text_value = child.text or None
|
|
1227
1100
|
if tag not in by_full:
|
|
1228
1101
|
by_full[tag] = text_value
|
|
1229
1102
|
local = tag.rsplit("}", 1)[-1].lower()
|
|
@@ -1245,8 +1118,6 @@ def _first_non_empty(mapping: dict[str, Optional[str]], keys: tuple[str, ...]) -
|
|
|
1245
1118
|
def _parse_rss_feed_entry_fast(
|
|
1246
1119
|
item: _Element,
|
|
1247
1120
|
atom_ns: str,
|
|
1248
|
-
parse_media_content: bool = True,
|
|
1249
|
-
parse_enclosures: bool = True,
|
|
1250
1121
|
) -> FastFeedParserDict:
|
|
1251
1122
|
text_by_local, text_by_full = _build_rss_item_text_maps(item)
|
|
1252
1123
|
|
|
@@ -1268,7 +1139,7 @@ def _parse_rss_feed_entry_fast(
|
|
|
1268
1139
|
|
|
1269
1140
|
link = text_by_local.get("link")
|
|
1270
1141
|
if link:
|
|
1271
|
-
entry["link"] = link
|
|
1142
|
+
entry["link"] = link.strip()
|
|
1272
1143
|
|
|
1273
1144
|
published_source = _first_non_empty(text_by_local, ("pubdate", "published", "issued", "date"))
|
|
1274
1145
|
if published_source:
|
|
@@ -1282,7 +1153,7 @@ def _parse_rss_feed_entry_fast(
|
|
|
1282
1153
|
if updated:
|
|
1283
1154
|
entry["updated"] = updated
|
|
1284
1155
|
|
|
1285
|
-
if "published" not in entry and rss_guid:
|
|
1156
|
+
if "published" not in entry and rss_guid and not rss_guid.startswith(("http://", "https://")):
|
|
1286
1157
|
guid_date = _parse_date(rss_guid)
|
|
1287
1158
|
if guid_date:
|
|
1288
1159
|
entry["published"] = guid_date
|
|
@@ -1296,26 +1167,24 @@ def _parse_rss_feed_entry_fast(
|
|
|
1296
1167
|
|
|
1297
1168
|
_populate_entry_content(entry, item, "rss", atom_ns)
|
|
1298
1169
|
|
|
1299
|
-
|
|
1300
|
-
|
|
1301
|
-
|
|
1302
|
-
entry["media_content"] = media_contents
|
|
1170
|
+
media_contents = _parse_media_content(item)
|
|
1171
|
+
if media_contents:
|
|
1172
|
+
entry["media_content"] = media_contents
|
|
1303
1173
|
|
|
1304
|
-
|
|
1305
|
-
|
|
1306
|
-
|
|
1307
|
-
entry["enclosures"] = enclosures
|
|
1174
|
+
enclosures = _parse_enclosures(item)
|
|
1175
|
+
if enclosures:
|
|
1176
|
+
entry["enclosures"] = enclosures
|
|
1308
1177
|
|
|
1309
1178
|
author = _first_non_empty(text_by_local, ("author", "creator"))
|
|
1310
1179
|
if not author:
|
|
1311
1180
|
atom_author = item.find(f"{{{atom_ns}}}author/{{{atom_ns}}}name")
|
|
1312
1181
|
author = atom_author.text.strip() if atom_author is not None and atom_author.text else None
|
|
1313
1182
|
if author:
|
|
1314
|
-
entry["author"] = author
|
|
1183
|
+
entry["author"] = author.strip()
|
|
1315
1184
|
|
|
1316
1185
|
comments = text_by_local.get("comments")
|
|
1317
1186
|
if comments:
|
|
1318
|
-
entry["comments"] = comments
|
|
1187
|
+
entry["comments"] = comments.strip()
|
|
1319
1188
|
|
|
1320
1189
|
tags = _parse_tags(item, "rss", atom_ns)
|
|
1321
1190
|
if tags:
|
|
@@ -1324,25 +1193,114 @@ def _parse_rss_feed_entry_fast(
|
|
|
1324
1193
|
return entry
|
|
1325
1194
|
|
|
1326
1195
|
|
|
1196
|
+
def _parse_atom_feed_entry_fast(
|
|
1197
|
+
item: _Element,
|
|
1198
|
+
atom_ns: str,
|
|
1199
|
+
) -> FastFeedParserDict:
|
|
1200
|
+
ns = f"{{{atom_ns}}}"
|
|
1201
|
+
entry = FastFeedParserDict()
|
|
1202
|
+
|
|
1203
|
+
# ID
|
|
1204
|
+
el = item.find(ns + "id")
|
|
1205
|
+
if el is not None and el.text:
|
|
1206
|
+
entry["id"] = el.text.strip()
|
|
1207
|
+
|
|
1208
|
+
# Title
|
|
1209
|
+
el = item.find(ns + "title")
|
|
1210
|
+
if el is not None and el.text:
|
|
1211
|
+
entry["title"] = el.text.strip()
|
|
1212
|
+
|
|
1213
|
+
# Description (summary)
|
|
1214
|
+
el = item.find(ns + "summary")
|
|
1215
|
+
if el is not None and el.text:
|
|
1216
|
+
entry["description"] = el.text.strip()
|
|
1217
|
+
|
|
1218
|
+
# Link (href attribute)
|
|
1219
|
+
el = item.find(ns + "link")
|
|
1220
|
+
if el is not None:
|
|
1221
|
+
href = el.get("href")
|
|
1222
|
+
if href:
|
|
1223
|
+
entry["link"] = href.strip()
|
|
1224
|
+
|
|
1225
|
+
# Dates: Atom 1.0 uses published/updated, Atom 0.3 uses issued/modified
|
|
1226
|
+
is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
|
|
1227
|
+
pub_tag = "issued" if is_atom_03 else "published"
|
|
1228
|
+
upd_tag = "modified" if is_atom_03 else "updated"
|
|
1229
|
+
pub_fallback_tag = "published" if is_atom_03 else "issued"
|
|
1230
|
+
upd_fallback_tag = "updated" if is_atom_03 else "modified"
|
|
1231
|
+
|
|
1232
|
+
el = item.find(ns + pub_tag)
|
|
1233
|
+
if el is not None and el.text:
|
|
1234
|
+
published = _parse_date(el.text)
|
|
1235
|
+
if published:
|
|
1236
|
+
entry["published"] = published
|
|
1237
|
+
|
|
1238
|
+
el = item.find(ns + upd_tag)
|
|
1239
|
+
if el is not None and el.text:
|
|
1240
|
+
updated = _parse_date(el.text)
|
|
1241
|
+
if updated:
|
|
1242
|
+
entry["updated"] = updated
|
|
1243
|
+
|
|
1244
|
+
# Fallback date fields for mixed namespace scenarios
|
|
1245
|
+
if "published" not in entry:
|
|
1246
|
+
el = item.find(ns + pub_fallback_tag)
|
|
1247
|
+
if el is not None and el.text:
|
|
1248
|
+
published = _parse_date(el.text)
|
|
1249
|
+
if published:
|
|
1250
|
+
entry["published"] = published
|
|
1251
|
+
|
|
1252
|
+
if "updated" not in entry:
|
|
1253
|
+
el = item.find(ns + upd_fallback_tag)
|
|
1254
|
+
if el is not None and el.text:
|
|
1255
|
+
updated = _parse_date(el.text)
|
|
1256
|
+
if updated:
|
|
1257
|
+
entry["updated"] = updated
|
|
1258
|
+
|
|
1259
|
+
if "updated" in entry and "published" not in entry:
|
|
1260
|
+
entry["published"] = entry["updated"]
|
|
1261
|
+
|
|
1262
|
+
_populate_entry_links(entry, item, atom_ns)
|
|
1263
|
+
|
|
1264
|
+
if "id" not in entry and "link" in entry:
|
|
1265
|
+
entry["id"] = entry["link"]
|
|
1266
|
+
|
|
1267
|
+
_populate_entry_content(entry, item, "atom", atom_ns)
|
|
1268
|
+
|
|
1269
|
+
media_contents = _parse_media_content(item)
|
|
1270
|
+
if media_contents:
|
|
1271
|
+
entry["media_content"] = media_contents
|
|
1272
|
+
|
|
1273
|
+
enclosures = _parse_enclosures(item)
|
|
1274
|
+
if enclosures:
|
|
1275
|
+
entry["enclosures"] = enclosures
|
|
1276
|
+
|
|
1277
|
+
# Author
|
|
1278
|
+
el = item.find(ns + "author/" + ns + "name")
|
|
1279
|
+
if el is not None and el.text:
|
|
1280
|
+
entry["author"] = el.text.strip()
|
|
1281
|
+
|
|
1282
|
+
tags = _parse_tags(item, "atom", atom_ns)
|
|
1283
|
+
if tags:
|
|
1284
|
+
entry["tags"] = tags
|
|
1285
|
+
|
|
1286
|
+
return entry
|
|
1287
|
+
|
|
1288
|
+
|
|
1327
1289
|
def _parse_feed_entry(
|
|
1328
1290
|
item: _Element,
|
|
1329
1291
|
feed_type: _FeedType,
|
|
1330
1292
|
atom_namespace: Optional[str] = None,
|
|
1331
|
-
*,
|
|
1332
|
-
parse_media_content: bool = True,
|
|
1333
|
-
parse_enclosures: bool = True,
|
|
1334
1293
|
) -> FastFeedParserDict:
|
|
1335
1294
|
# Use dynamic atom namespace or fallback to default
|
|
1336
1295
|
atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
|
|
1337
1296
|
|
|
1338
1297
|
if feed_type == "rss":
|
|
1339
|
-
return _parse_rss_feed_entry_fast(
|
|
1340
|
-
item,
|
|
1341
|
-
atom_ns,
|
|
1342
|
-
parse_media_content=parse_media_content,
|
|
1343
|
-
parse_enclosures=parse_enclosures,
|
|
1344
|
-
)
|
|
1298
|
+
return _parse_rss_feed_entry_fast(item, atom_ns)
|
|
1345
1299
|
|
|
1300
|
+
if feed_type == "atom":
|
|
1301
|
+
return _parse_atom_feed_entry_fast(item, atom_ns)
|
|
1302
|
+
|
|
1303
|
+
# RDF path uses the generic field machinery
|
|
1346
1304
|
# Check if this is Atom 0.3 to use different date field names
|
|
1347
1305
|
is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
|
|
1348
1306
|
|
|
@@ -1434,8 +1392,7 @@ def _parse_feed_entry(
|
|
|
1434
1392
|
entry["updated"] = _parse_date(fallback_updated)
|
|
1435
1393
|
|
|
1436
1394
|
# Try to extract date from GUID as final fallback
|
|
1437
|
-
if "published" not in entry and rss_guid:
|
|
1438
|
-
# Check if GUID contains date information
|
|
1395
|
+
if "published" not in entry and rss_guid and not rss_guid.startswith(("http://", "https://")):
|
|
1439
1396
|
guid_date = _parse_date(rss_guid)
|
|
1440
1397
|
if guid_date:
|
|
1441
1398
|
entry["published"] = guid_date
|
|
@@ -1455,15 +1412,13 @@ def _parse_feed_entry(
|
|
|
1455
1412
|
|
|
1456
1413
|
_populate_entry_content(entry, item, feed_type, atom_ns)
|
|
1457
1414
|
|
|
1458
|
-
|
|
1459
|
-
|
|
1460
|
-
|
|
1461
|
-
entry["media_content"] = media_contents
|
|
1415
|
+
media_contents = _parse_media_content(item)
|
|
1416
|
+
if media_contents:
|
|
1417
|
+
entry["media_content"] = media_contents
|
|
1462
1418
|
|
|
1463
|
-
|
|
1464
|
-
|
|
1465
|
-
|
|
1466
|
-
entry["enclosures"] = enclosures
|
|
1419
|
+
enclosures = _parse_enclosures(item)
|
|
1420
|
+
if enclosures:
|
|
1421
|
+
entry["enclosures"] = enclosures
|
|
1467
1422
|
|
|
1468
1423
|
author = get_field_value(
|
|
1469
1424
|
"author",
|
|
@@ -1618,6 +1573,14 @@ def _normalize_iso_datetime_string(value: str) -> str:
|
|
|
1618
1573
|
if not cleaned:
|
|
1619
1574
|
return cleaned
|
|
1620
1575
|
|
|
1576
|
+
# Fast path: 'Z' suffix (most common in Atom feeds)
|
|
1577
|
+
if cleaned[-1] in ("Z", "z"):
|
|
1578
|
+
return cleaned[:-1] + "+00:00"
|
|
1579
|
+
|
|
1580
|
+
# Fast path: already has proper +HH:MM or -HH:MM timezone
|
|
1581
|
+
if len(cleaned) > 6 and cleaned[-6] in ("+", "-") and cleaned[-3] == ":":
|
|
1582
|
+
return cleaned
|
|
1583
|
+
|
|
1621
1584
|
upper_cleaned = cleaned.upper()
|
|
1622
1585
|
for suffix in (" UTC", " GMT", " Z"):
|
|
1623
1586
|
if upper_cleaned.endswith(suffix):
|
|
@@ -1664,7 +1627,7 @@ def _parsedate_to_utc(value: str) -> Optional[datetime.datetime]:
|
|
|
1664
1627
|
return _ensure_utc(parsed)
|
|
1665
1628
|
|
|
1666
1629
|
|
|
1667
|
-
|
|
1630
|
+
_custom_tzinfos: dict[str, int] = {
|
|
1668
1631
|
"UTC": 0,
|
|
1669
1632
|
"UT": 0,
|
|
1670
1633
|
"GMT": 0,
|
|
@@ -1715,7 +1678,7 @@ _DATEPARSER_SETTINGS = {
|
|
|
1715
1678
|
@lru_cache(maxsize=512)
|
|
1716
1679
|
def _slow_dateutil_parse(value: str) -> Optional[datetime.datetime]:
|
|
1717
1680
|
try:
|
|
1718
|
-
return dateutil_parser.parse(value, tzinfos=
|
|
1681
|
+
return dateutil_parser.parse(value, tzinfos=_custom_tzinfos, ignoretz=False)
|
|
1719
1682
|
except (ValueError, TypeError, OverflowError):
|
|
1720
1683
|
return None
|
|
1721
1684
|
|
|
@@ -1727,7 +1690,7 @@ def _slow_dateparser(value: str) -> Optional[datetime.datetime]:
|
|
|
1727
1690
|
except ImportError:
|
|
1728
1691
|
return None
|
|
1729
1692
|
try:
|
|
1730
|
-
return _dateparser.parse(value, languages=["en"], settings=_DATEPARSER_SETTINGS)
|
|
1693
|
+
return _dateparser.parse(value, languages=["en"], settings={**_DATEPARSER_SETTINGS})
|
|
1731
1694
|
except (ValueError, TypeError):
|
|
1732
1695
|
return None
|
|
1733
1696
|
|
|
@@ -1747,6 +1710,27 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1747
1710
|
candidate = date_str.strip()
|
|
1748
1711
|
if not candidate:
|
|
1749
1712
|
return None
|
|
1713
|
+
|
|
1714
|
+
# Fast path: clean ISO-8601 (covers >90% of Atom/modern RSS dates)
|
|
1715
|
+
clen = len(candidate)
|
|
1716
|
+
if clen >= 20 and candidate[4] == "-" and candidate[0:4].isdigit():
|
|
1717
|
+
last = candidate[-1]
|
|
1718
|
+
# Most common: ends with 'Z' (e.g., 2024-01-15T10:30:00Z)
|
|
1719
|
+
if last in ("Z", "z"):
|
|
1720
|
+
try:
|
|
1721
|
+
dt = datetime.datetime.fromisoformat(candidate[:-1] + "+00:00")
|
|
1722
|
+
return dt.isoformat()
|
|
1723
|
+
except ValueError:
|
|
1724
|
+
pass # Fall through to full parsing
|
|
1725
|
+
# Second most common: ends with +HH:MM (e.g., 2024-01-15T10:30:00+00:00)
|
|
1726
|
+
elif clen > 6 and candidate[-6] in ("+", "-") and candidate[-3] == ":":
|
|
1727
|
+
try:
|
|
1728
|
+
dt = datetime.datetime.fromisoformat(candidate)
|
|
1729
|
+
utc_dt = dt.replace(tzinfo=_UTC) if dt.tzinfo is None else dt.astimezone(_UTC)
|
|
1730
|
+
return utc_dt.isoformat()
|
|
1731
|
+
except (ValueError, OverflowError):
|
|
1732
|
+
pass # Fall through to full parsing
|
|
1733
|
+
|
|
1750
1734
|
if "\n" in candidate or "\r" in candidate or "\t" in candidate or " " in candidate:
|
|
1751
1735
|
candidate = _RE_WHITESPACE.sub(" ", candidate)
|
|
1752
1736
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{fastfeedparser-0.5.0 → fastfeedparser-0.5.2}/src/fastfeedparser.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|