fastfeedparser 0.5.2__tar.gz → 0.5.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fastfeedparser-0.5.2/src/fastfeedparser.egg-info → fastfeedparser-0.5.3}/PKG-INFO +1 -1
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.3}/setup.cfg +1 -1
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.3}/src/fastfeedparser/main.py +134 -37
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.3/src/fastfeedparser.egg-info}/PKG-INFO +1 -1
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.3}/LICENSE +0 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.3}/README.md +0 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.3}/pyproject.toml +0 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.3}/src/fastfeedparser/__init__.py +0 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.3}/src/fastfeedparser.egg-info/SOURCES.txt +0 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.3}/src/fastfeedparser.egg-info/dependency_links.txt +0 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.3}/src/fastfeedparser.egg-info/requires.txt +0 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.3}/src/fastfeedparser.egg-info/top_level.txt +0 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.3}/tests/test_encoding.py +0 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.3}/tests/test_integration.py +0 -0
|
@@ -56,6 +56,18 @@ _RE_WHITESPACE = re.compile(r"\s+")
|
|
|
56
56
|
_RE_ISO_TZ_NO_COLON = re.compile(r"([+-]\d{2})(\d{2})$")
|
|
57
57
|
_RE_ISO_TZ_HOUR_ONLY = re.compile(r"([+-]\d{2})$")
|
|
58
58
|
_RE_ISO_FRACTION = re.compile(r"\.(\d{7,})(?=(?:[+-]\d{2}:?\d{2}|Z|$))", re.IGNORECASE)
|
|
59
|
+
_RE_RFC822 = re.compile(
|
|
60
|
+
r"(?:\w{3},\s+)?(\d{1,2})\s+(\w{3})\s+(\d{4})\s+(\d{2}):(\d{2}):(\d{2})\s+([+-]\d{4}|[A-Z]{2,5})"
|
|
61
|
+
)
|
|
62
|
+
_MONTHS_RFC822: dict[str, int] = {
|
|
63
|
+
"jan": 1, "feb": 2, "mar": 3, "apr": 4, "may": 5, "jun": 6,
|
|
64
|
+
"jul": 7, "aug": 8, "sep": 9, "oct": 10, "nov": 11, "dec": 12,
|
|
65
|
+
}
|
|
66
|
+
_TZ_OFFSETS_RFC822: dict[str, int] = {
|
|
67
|
+
"GMT": 0, "UTC": 0, "UT": 0,
|
|
68
|
+
"EST": -18000, "EDT": -14400, "CST": -21600, "CDT": -18000,
|
|
69
|
+
"MST": -25200, "MDT": -21600, "PST": -28800, "PDT": -25200,
|
|
70
|
+
}
|
|
59
71
|
|
|
60
72
|
|
|
61
73
|
class FastFeedParserDict(dict):
|
|
@@ -382,24 +394,26 @@ def _maybe_parse_json_feed(content: str | bytes) -> FastFeedParserDict | None:
|
|
|
382
394
|
return None
|
|
383
395
|
|
|
384
396
|
|
|
397
|
+
_STRICT_XML_PARSER = etree.XMLParser(
|
|
398
|
+
ns_clean=True,
|
|
399
|
+
recover=False,
|
|
400
|
+
collect_ids=False,
|
|
401
|
+
resolve_entities=False,
|
|
402
|
+
)
|
|
403
|
+
_RECOVER_XML_PARSER = etree.XMLParser(
|
|
404
|
+
ns_clean=True,
|
|
405
|
+
recover=True,
|
|
406
|
+
collect_ids=False,
|
|
407
|
+
resolve_entities=False,
|
|
408
|
+
)
|
|
409
|
+
|
|
410
|
+
|
|
385
411
|
def _parse_xml_root(xml_content: bytes) -> _Element:
|
|
386
412
|
try:
|
|
387
|
-
|
|
388
|
-
ns_clean=True,
|
|
389
|
-
recover=False,
|
|
390
|
-
collect_ids=False,
|
|
391
|
-
resolve_entities=False,
|
|
392
|
-
)
|
|
393
|
-
root = etree.fromstring(xml_content, parser=strict_parser)
|
|
413
|
+
root = etree.fromstring(xml_content, parser=_STRICT_XML_PARSER)
|
|
394
414
|
except etree.XMLSyntaxError:
|
|
395
|
-
recover_parser = etree.XMLParser(
|
|
396
|
-
ns_clean=True,
|
|
397
|
-
recover=True,
|
|
398
|
-
collect_ids=False,
|
|
399
|
-
resolve_entities=False,
|
|
400
|
-
)
|
|
401
415
|
try:
|
|
402
|
-
root = etree.fromstring(xml_content, parser=
|
|
416
|
+
root = etree.fromstring(xml_content, parser=_RECOVER_XML_PARSER)
|
|
403
417
|
except etree.XMLSyntaxError as e:
|
|
404
418
|
raise ValueError(f"Failed to parse XML content: {str(e)}")
|
|
405
419
|
|
|
@@ -642,6 +656,9 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
|
642
656
|
|
|
643
657
|
feed = _parse_feed_info(channel, feed_type, atom_namespace)
|
|
644
658
|
|
|
659
|
+
# Detect once whether media namespace is used anywhere in the document
|
|
660
|
+
has_media_ns = b"search.yahoo.com/mrss" in xml_content if isinstance(xml_content, bytes) else "search.yahoo.com/mrss" in xml_content
|
|
661
|
+
|
|
645
662
|
# Parse entries
|
|
646
663
|
entries: list[FastFeedParserDict] = []
|
|
647
664
|
feed["entries"] = entries
|
|
@@ -650,6 +667,7 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
|
650
667
|
item,
|
|
651
668
|
feed_type,
|
|
652
669
|
atom_namespace,
|
|
670
|
+
has_media_ns,
|
|
653
671
|
)
|
|
654
672
|
# Ensure that titles and descriptions are always present
|
|
655
673
|
entry["title"] = entry.get("title", "").strip()
|
|
@@ -998,7 +1016,10 @@ def _populate_entry_content(
|
|
|
998
1016
|
if "<" in content_value:
|
|
999
1017
|
content_value = _RE_HTML_TAGS.sub(" ", content_value[:2048])
|
|
1000
1018
|
content_value = _html_mod.unescape(content_value)
|
|
1001
|
-
content_value
|
|
1019
|
+
if " " in content_value or "\n" in content_value or "\t" in content_value or "\r" in content_value:
|
|
1020
|
+
content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
|
|
1021
|
+
else:
|
|
1022
|
+
content_value = content_value.strip()
|
|
1002
1023
|
entry["description"] = content_value[:512]
|
|
1003
1024
|
|
|
1004
1025
|
|
|
@@ -1099,9 +1120,13 @@ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]],
|
|
|
1099
1120
|
text_value = child.text or None
|
|
1100
1121
|
if tag not in by_full:
|
|
1101
1122
|
by_full[tag] = text_value
|
|
1102
|
-
|
|
1103
|
-
if "
|
|
1104
|
-
local =
|
|
1123
|
+
# Fast path: ~80% of RSS tags have no namespace or colon prefix
|
|
1124
|
+
if "{" in tag:
|
|
1125
|
+
local = tag.rsplit("}", 1)[1].lower()
|
|
1126
|
+
elif ":" in tag:
|
|
1127
|
+
local = tag.split(":", 1)[1].lower()
|
|
1128
|
+
else:
|
|
1129
|
+
local = tag.lower()
|
|
1105
1130
|
if local not in by_local:
|
|
1106
1131
|
by_local[local] = text_value
|
|
1107
1132
|
return by_local, by_full
|
|
@@ -1118,6 +1143,7 @@ def _first_non_empty(mapping: dict[str, Optional[str]], keys: tuple[str, ...]) -
|
|
|
1118
1143
|
def _parse_rss_feed_entry_fast(
|
|
1119
1144
|
item: _Element,
|
|
1120
1145
|
atom_ns: str,
|
|
1146
|
+
has_media_ns: bool = True,
|
|
1121
1147
|
) -> FastFeedParserDict:
|
|
1122
1148
|
text_by_local, text_by_full = _build_rss_item_text_maps(item)
|
|
1123
1149
|
|
|
@@ -1161,15 +1187,26 @@ def _parse_rss_feed_entry_fast(
|
|
|
1161
1187
|
if "updated" in entry and "published" not in entry:
|
|
1162
1188
|
entry["published"] = entry["updated"]
|
|
1163
1189
|
|
|
1164
|
-
|
|
1190
|
+
# Inline link population for RSS (avoids redundant findall/find for 98.8% of entries)
|
|
1191
|
+
atom_links = item.findall(f"{{{atom_ns}}}link")
|
|
1192
|
+
if atom_links:
|
|
1193
|
+
# Has atom:link elements - use full logic
|
|
1194
|
+
_populate_entry_links(entry, item, atom_ns)
|
|
1195
|
+
else:
|
|
1196
|
+
# Common RSS case: no atom:link elements
|
|
1197
|
+
entry["links"] = []
|
|
1198
|
+
if "link" not in entry and rss_guid and rss_guid.startswith(("http://", "https://")):
|
|
1199
|
+
entry["link"] = rss_guid
|
|
1200
|
+
|
|
1165
1201
|
if "id" not in entry and "link" in entry:
|
|
1166
1202
|
entry["id"] = entry["link"]
|
|
1167
1203
|
|
|
1168
1204
|
_populate_entry_content(entry, item, "rss", atom_ns)
|
|
1169
1205
|
|
|
1170
|
-
|
|
1171
|
-
|
|
1172
|
-
|
|
1206
|
+
if has_media_ns:
|
|
1207
|
+
media_contents = _parse_media_content(item)
|
|
1208
|
+
if media_contents:
|
|
1209
|
+
entry["media_content"] = media_contents
|
|
1173
1210
|
|
|
1174
1211
|
enclosures = _parse_enclosures(item)
|
|
1175
1212
|
if enclosures:
|
|
@@ -1196,6 +1233,7 @@ def _parse_rss_feed_entry_fast(
|
|
|
1196
1233
|
def _parse_atom_feed_entry_fast(
|
|
1197
1234
|
item: _Element,
|
|
1198
1235
|
atom_ns: str,
|
|
1236
|
+
has_media_ns: bool = True,
|
|
1199
1237
|
) -> FastFeedParserDict:
|
|
1200
1238
|
ns = f"{{{atom_ns}}}"
|
|
1201
1239
|
entry = FastFeedParserDict()
|
|
@@ -1266,9 +1304,10 @@ def _parse_atom_feed_entry_fast(
|
|
|
1266
1304
|
|
|
1267
1305
|
_populate_entry_content(entry, item, "atom", atom_ns)
|
|
1268
1306
|
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
|
|
1307
|
+
if has_media_ns:
|
|
1308
|
+
media_contents = _parse_media_content(item)
|
|
1309
|
+
if media_contents:
|
|
1310
|
+
entry["media_content"] = media_contents
|
|
1272
1311
|
|
|
1273
1312
|
enclosures = _parse_enclosures(item)
|
|
1274
1313
|
if enclosures:
|
|
@@ -1290,15 +1329,16 @@ def _parse_feed_entry(
|
|
|
1290
1329
|
item: _Element,
|
|
1291
1330
|
feed_type: _FeedType,
|
|
1292
1331
|
atom_namespace: Optional[str] = None,
|
|
1332
|
+
has_media_ns: bool = True,
|
|
1293
1333
|
) -> FastFeedParserDict:
|
|
1294
1334
|
# Use dynamic atom namespace or fallback to default
|
|
1295
1335
|
atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
|
|
1296
1336
|
|
|
1297
1337
|
if feed_type == "rss":
|
|
1298
|
-
return _parse_rss_feed_entry_fast(item, atom_ns)
|
|
1338
|
+
return _parse_rss_feed_entry_fast(item, atom_ns, has_media_ns)
|
|
1299
1339
|
|
|
1300
1340
|
if feed_type == "atom":
|
|
1301
|
-
return _parse_atom_feed_entry_fast(item, atom_ns)
|
|
1341
|
+
return _parse_atom_feed_entry_fast(item, atom_ns, has_media_ns)
|
|
1302
1342
|
|
|
1303
1343
|
# RDF path uses the generic field machinery
|
|
1304
1344
|
# Check if this is Atom 0.3 to use different date field names
|
|
@@ -1412,9 +1452,10 @@ def _parse_feed_entry(
|
|
|
1412
1452
|
|
|
1413
1453
|
_populate_entry_content(entry, item, feed_type, atom_ns)
|
|
1414
1454
|
|
|
1415
|
-
|
|
1416
|
-
|
|
1417
|
-
|
|
1455
|
+
if has_media_ns:
|
|
1456
|
+
media_contents = _parse_media_content(item)
|
|
1457
|
+
if media_contents:
|
|
1458
|
+
entry["media_content"] = media_contents
|
|
1418
1459
|
|
|
1419
1460
|
enclosures = _parse_enclosures(item)
|
|
1420
1461
|
if enclosures:
|
|
@@ -1616,8 +1657,54 @@ def _ensure_utc(dt: datetime.datetime) -> Optional[datetime.datetime]:
|
|
|
1616
1657
|
return None
|
|
1617
1658
|
|
|
1618
1659
|
|
|
1660
|
+
def _fast_rfc822_to_iso(value: str) -> Optional[str]:
|
|
1661
|
+
"""Fast RFC-822 date to ISO string, bypassing datetime objects for UTC dates."""
|
|
1662
|
+
m = _RE_RFC822.match(value)
|
|
1663
|
+
if not m:
|
|
1664
|
+
return None
|
|
1665
|
+
day, mon_str, year, hour, minute, second, tz = m.groups()
|
|
1666
|
+
month = _MONTHS_RFC822.get(mon_str.lower())
|
|
1667
|
+
if month is None:
|
|
1668
|
+
return None
|
|
1669
|
+
if tz[0] in "+-":
|
|
1670
|
+
tz_offset_seconds = (int(tz[1:3]) * 3600 + int(tz[3:5]) * 60) * (
|
|
1671
|
+
1 if tz[0] == "+" else -1
|
|
1672
|
+
)
|
|
1673
|
+
else:
|
|
1674
|
+
tz_offset_seconds = _TZ_OFFSETS_RFC822.get(tz)
|
|
1675
|
+
if tz_offset_seconds is None:
|
|
1676
|
+
return None # Unknown tz name, fall through to full parser
|
|
1677
|
+
# Python requires offset strictly between -24h and +24h
|
|
1678
|
+
if not (-86400 < tz_offset_seconds < 86400):
|
|
1679
|
+
return None
|
|
1680
|
+
d = int(day)
|
|
1681
|
+
h = int(hour)
|
|
1682
|
+
mi = int(minute)
|
|
1683
|
+
s = int(second)
|
|
1684
|
+
# Hour 24 is invalid (even ISO only allows 24:00:00); roll to next day at 00:mm:ss
|
|
1685
|
+
if h == 24:
|
|
1686
|
+
base = datetime.date(int(year), month, d) + datetime.timedelta(days=1)
|
|
1687
|
+
h = 0
|
|
1688
|
+
if tz_offset_seconds == 0:
|
|
1689
|
+
return f"{base.year:04d}-{base.month:02d}-{base.day:02d}T{h:02d}:{mi:02d}:{s:02d}+00:00"
|
|
1690
|
+
dt = datetime.datetime(
|
|
1691
|
+
base.year, base.month, base.day, h, mi, s,
|
|
1692
|
+
tzinfo=datetime.timezone(datetime.timedelta(seconds=tz_offset_seconds)),
|
|
1693
|
+
)
|
|
1694
|
+
utc = dt.astimezone(_UTC)
|
|
1695
|
+
return f"{utc.year:04d}-{utc.month:02d}-{utc.day:02d}T{utc.hour:02d}:{utc.minute:02d}:{utc.second:02d}+00:00"
|
|
1696
|
+
if tz_offset_seconds == 0:
|
|
1697
|
+
return f"{year}-{month:02d}-{d:02d}T{hour}:{minute}:{second}+00:00"
|
|
1698
|
+
dt = datetime.datetime(
|
|
1699
|
+
int(year), month, d, h, mi, s,
|
|
1700
|
+
tzinfo=datetime.timezone(datetime.timedelta(seconds=tz_offset_seconds)),
|
|
1701
|
+
)
|
|
1702
|
+
utc = dt.astimezone(_UTC)
|
|
1703
|
+
return f"{utc.year:04d}-{utc.month:02d}-{utc.day:02d}T{utc.hour:02d}:{utc.minute:02d}:{utc.second:02d}+00:00"
|
|
1704
|
+
|
|
1705
|
+
|
|
1619
1706
|
def _parsedate_to_utc(value: str) -> Optional[datetime.datetime]:
|
|
1620
|
-
"""
|
|
1707
|
+
"""RFC-822 / RFC-2822 parsing via email.utils (fallback)."""
|
|
1621
1708
|
try:
|
|
1622
1709
|
parsed = parsedate_to_datetime(value)
|
|
1623
1710
|
except (TypeError, ValueError, IndexError):
|
|
@@ -1717,8 +1804,9 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1717
1804
|
last = candidate[-1]
|
|
1718
1805
|
# Most common: ends with 'Z' (e.g., 2024-01-15T10:30:00Z)
|
|
1719
1806
|
if last in ("Z", "z"):
|
|
1807
|
+
iso = candidate[:-1] + "+00:00"
|
|
1720
1808
|
try:
|
|
1721
|
-
dt = datetime.datetime.fromisoformat(
|
|
1809
|
+
dt = datetime.datetime.fromisoformat(iso)
|
|
1722
1810
|
return dt.isoformat()
|
|
1723
1811
|
except ValueError:
|
|
1724
1812
|
pass # Fall through to full parsing
|
|
@@ -1726,7 +1814,9 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1726
1814
|
elif clen > 6 and candidate[-6] in ("+", "-") and candidate[-3] == ":":
|
|
1727
1815
|
try:
|
|
1728
1816
|
dt = datetime.datetime.fromisoformat(candidate)
|
|
1729
|
-
|
|
1817
|
+
if dt.tzinfo is _UTC:
|
|
1818
|
+
return dt.isoformat()
|
|
1819
|
+
utc_dt = dt.astimezone(_UTC)
|
|
1730
1820
|
return utc_dt.isoformat()
|
|
1731
1821
|
except (ValueError, OverflowError):
|
|
1732
1822
|
pass # Fall through to full parsing
|
|
@@ -1743,10 +1833,13 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1743
1833
|
if not ((year % 4 == 0 and year % 100 != 0) or (year % 400 == 0)):
|
|
1744
1834
|
candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
|
|
1745
1835
|
|
|
1746
|
-
if "24:
|
|
1747
|
-
|
|
1748
|
-
|
|
1749
|
-
|
|
1836
|
+
if "T24:" in candidate or " 24:" in candidate:
|
|
1837
|
+
m24 = re.search(r"(\d{4}-\d{2}-\d{2})[T ]24:(\d{2}):(\d{2})", candidate)
|
|
1838
|
+
if m24:
|
|
1839
|
+
base = datetime.date.fromisoformat(m24.group(1))
|
|
1840
|
+
mins, secs = int(m24.group(2)), int(m24.group(3))
|
|
1841
|
+
next_day = base + datetime.timedelta(days=1)
|
|
1842
|
+
candidate = candidate[:m24.start()] + f"{next_day}T00:{mins:02d}:{secs:02d}" + candidate[m24.end():]
|
|
1750
1843
|
|
|
1751
1844
|
dt: Optional[datetime.datetime] = None
|
|
1752
1845
|
|
|
@@ -1762,6 +1855,10 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1762
1855
|
if utc_dt is not None:
|
|
1763
1856
|
return utc_dt.isoformat()
|
|
1764
1857
|
|
|
1858
|
+
rfc822_result = _fast_rfc822_to_iso(candidate)
|
|
1859
|
+
if rfc822_result is not None:
|
|
1860
|
+
return rfc822_result
|
|
1861
|
+
|
|
1765
1862
|
dt = _parsedate_to_utc(candidate)
|
|
1766
1863
|
if dt is not None:
|
|
1767
1864
|
return dt.isoformat()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{fastfeedparser-0.5.2 → fastfeedparser-0.5.3}/src/fastfeedparser.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|