fastfeedparser 0.5.1__tar.gz → 0.5.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.1
3
+ Version: 0.5.3
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = fastfeedparser
3
- version = 0.5.1
3
+ version = 0.5.3
4
4
  author = Vladimir Prelovac
5
5
  author_email = vlad@kagi.com
6
6
  description = High performance RSS, Atom, JSON and RDF feed parser in Python
@@ -56,6 +56,18 @@ _RE_WHITESPACE = re.compile(r"\s+")
56
56
  _RE_ISO_TZ_NO_COLON = re.compile(r"([+-]\d{2})(\d{2})$")
57
57
  _RE_ISO_TZ_HOUR_ONLY = re.compile(r"([+-]\d{2})$")
58
58
  _RE_ISO_FRACTION = re.compile(r"\.(\d{7,})(?=(?:[+-]\d{2}:?\d{2}|Z|$))", re.IGNORECASE)
59
+ _RE_RFC822 = re.compile(
60
+ r"(?:\w{3},\s+)?(\d{1,2})\s+(\w{3})\s+(\d{4})\s+(\d{2}):(\d{2}):(\d{2})\s+([+-]\d{4}|[A-Z]{2,5})"
61
+ )
62
+ _MONTHS_RFC822: dict[str, int] = {
63
+ "jan": 1, "feb": 2, "mar": 3, "apr": 4, "may": 5, "jun": 6,
64
+ "jul": 7, "aug": 8, "sep": 9, "oct": 10, "nov": 11, "dec": 12,
65
+ }
66
+ _TZ_OFFSETS_RFC822: dict[str, int] = {
67
+ "GMT": 0, "UTC": 0, "UT": 0,
68
+ "EST": -18000, "EDT": -14400, "CST": -21600, "CDT": -18000,
69
+ "MST": -25200, "MDT": -21600, "PST": -28800, "PDT": -25200,
70
+ }
59
71
 
60
72
 
61
73
  class FastFeedParserDict(dict):
@@ -382,24 +394,26 @@ def _maybe_parse_json_feed(content: str | bytes) -> FastFeedParserDict | None:
382
394
  return None
383
395
 
384
396
 
397
+ _STRICT_XML_PARSER = etree.XMLParser(
398
+ ns_clean=True,
399
+ recover=False,
400
+ collect_ids=False,
401
+ resolve_entities=False,
402
+ )
403
+ _RECOVER_XML_PARSER = etree.XMLParser(
404
+ ns_clean=True,
405
+ recover=True,
406
+ collect_ids=False,
407
+ resolve_entities=False,
408
+ )
409
+
410
+
385
411
  def _parse_xml_root(xml_content: bytes) -> _Element:
386
412
  try:
387
- strict_parser = etree.XMLParser(
388
- ns_clean=True,
389
- recover=False,
390
- collect_ids=False,
391
- resolve_entities=False,
392
- )
393
- root = etree.fromstring(xml_content, parser=strict_parser)
413
+ root = etree.fromstring(xml_content, parser=_STRICT_XML_PARSER)
394
414
  except etree.XMLSyntaxError:
395
- recover_parser = etree.XMLParser(
396
- ns_clean=True,
397
- recover=True,
398
- collect_ids=False,
399
- resolve_entities=False,
400
- )
401
415
  try:
402
- root = etree.fromstring(xml_content, parser=recover_parser)
416
+ root = etree.fromstring(xml_content, parser=_RECOVER_XML_PARSER)
403
417
  except etree.XMLSyntaxError as e:
404
418
  raise ValueError(f"Failed to parse XML content: {str(e)}")
405
419
 
@@ -642,6 +656,9 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
642
656
 
643
657
  feed = _parse_feed_info(channel, feed_type, atom_namespace)
644
658
 
659
+ # Detect once whether media namespace is used anywhere in the document
660
+ has_media_ns = b"search.yahoo.com/mrss" in xml_content if isinstance(xml_content, bytes) else "search.yahoo.com/mrss" in xml_content
661
+
645
662
  # Parse entries
646
663
  entries: list[FastFeedParserDict] = []
647
664
  feed["entries"] = entries
@@ -650,6 +667,7 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
650
667
  item,
651
668
  feed_type,
652
669
  atom_namespace,
670
+ has_media_ns,
653
671
  )
654
672
  # Ensure that titles and descriptions are always present
655
673
  entry["title"] = entry.get("title", "").strip()
@@ -998,7 +1016,10 @@ def _populate_entry_content(
998
1016
  if "<" in content_value:
999
1017
  content_value = _RE_HTML_TAGS.sub(" ", content_value[:2048])
1000
1018
  content_value = _html_mod.unescape(content_value)
1001
- content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
1019
+ if " " in content_value or "\n" in content_value or "\t" in content_value or "\r" in content_value:
1020
+ content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
1021
+ else:
1022
+ content_value = content_value.strip()
1002
1023
  entry["description"] = content_value[:512]
1003
1024
 
1004
1025
 
@@ -1096,12 +1117,16 @@ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]],
1096
1117
  tag = child.tag
1097
1118
  if not isinstance(tag, str):
1098
1119
  continue
1099
- text_value = child.text.strip() if child.text else None
1120
+ text_value = child.text or None
1100
1121
  if tag not in by_full:
1101
1122
  by_full[tag] = text_value
1102
- local = tag.rsplit("}", 1)[-1].lower()
1103
- if ":" in local:
1104
- local = local.split(":", 1)[1]
1123
+ # Fast path: ~80% of RSS tags have no namespace or colon prefix
1124
+ if "{" in tag:
1125
+ local = tag.rsplit("}", 1)[1].lower()
1126
+ elif ":" in tag:
1127
+ local = tag.split(":", 1)[1].lower()
1128
+ else:
1129
+ local = tag.lower()
1105
1130
  if local not in by_local:
1106
1131
  by_local[local] = text_value
1107
1132
  return by_local, by_full
@@ -1118,6 +1143,7 @@ def _first_non_empty(mapping: dict[str, Optional[str]], keys: tuple[str, ...]) -
1118
1143
  def _parse_rss_feed_entry_fast(
1119
1144
  item: _Element,
1120
1145
  atom_ns: str,
1146
+ has_media_ns: bool = True,
1121
1147
  ) -> FastFeedParserDict:
1122
1148
  text_by_local, text_by_full = _build_rss_item_text_maps(item)
1123
1149
 
@@ -1139,7 +1165,7 @@ def _parse_rss_feed_entry_fast(
1139
1165
 
1140
1166
  link = text_by_local.get("link")
1141
1167
  if link:
1142
- entry["link"] = link
1168
+ entry["link"] = link.strip()
1143
1169
 
1144
1170
  published_source = _first_non_empty(text_by_local, ("pubdate", "published", "issued", "date"))
1145
1171
  if published_source:
@@ -1161,15 +1187,26 @@ def _parse_rss_feed_entry_fast(
1161
1187
  if "updated" in entry and "published" not in entry:
1162
1188
  entry["published"] = entry["updated"]
1163
1189
 
1164
- _populate_entry_links(entry, item, atom_ns)
1190
+ # Inline link population for RSS (avoids redundant findall/find for 98.8% of entries)
1191
+ atom_links = item.findall(f"{{{atom_ns}}}link")
1192
+ if atom_links:
1193
+ # Has atom:link elements - use full logic
1194
+ _populate_entry_links(entry, item, atom_ns)
1195
+ else:
1196
+ # Common RSS case: no atom:link elements
1197
+ entry["links"] = []
1198
+ if "link" not in entry and rss_guid and rss_guid.startswith(("http://", "https://")):
1199
+ entry["link"] = rss_guid
1200
+
1165
1201
  if "id" not in entry and "link" in entry:
1166
1202
  entry["id"] = entry["link"]
1167
1203
 
1168
1204
  _populate_entry_content(entry, item, "rss", atom_ns)
1169
1205
 
1170
- media_contents = _parse_media_content(item)
1171
- if media_contents:
1172
- entry["media_content"] = media_contents
1206
+ if has_media_ns:
1207
+ media_contents = _parse_media_content(item)
1208
+ if media_contents:
1209
+ entry["media_content"] = media_contents
1173
1210
 
1174
1211
  enclosures = _parse_enclosures(item)
1175
1212
  if enclosures:
@@ -1180,11 +1217,11 @@ def _parse_rss_feed_entry_fast(
1180
1217
  atom_author = item.find(f"{{{atom_ns}}}author/{{{atom_ns}}}name")
1181
1218
  author = atom_author.text.strip() if atom_author is not None and atom_author.text else None
1182
1219
  if author:
1183
- entry["author"] = author
1220
+ entry["author"] = author.strip()
1184
1221
 
1185
1222
  comments = text_by_local.get("comments")
1186
1223
  if comments:
1187
- entry["comments"] = comments
1224
+ entry["comments"] = comments.strip()
1188
1225
 
1189
1226
  tags = _parse_tags(item, "rss", atom_ns)
1190
1227
  if tags:
@@ -1193,17 +1230,117 @@ def _parse_rss_feed_entry_fast(
1193
1230
  return entry
1194
1231
 
1195
1232
 
1233
+ def _parse_atom_feed_entry_fast(
1234
+ item: _Element,
1235
+ atom_ns: str,
1236
+ has_media_ns: bool = True,
1237
+ ) -> FastFeedParserDict:
1238
+ ns = f"{{{atom_ns}}}"
1239
+ entry = FastFeedParserDict()
1240
+
1241
+ # ID
1242
+ el = item.find(ns + "id")
1243
+ if el is not None and el.text:
1244
+ entry["id"] = el.text.strip()
1245
+
1246
+ # Title
1247
+ el = item.find(ns + "title")
1248
+ if el is not None and el.text:
1249
+ entry["title"] = el.text.strip()
1250
+
1251
+ # Description (summary)
1252
+ el = item.find(ns + "summary")
1253
+ if el is not None and el.text:
1254
+ entry["description"] = el.text.strip()
1255
+
1256
+ # Link (href attribute)
1257
+ el = item.find(ns + "link")
1258
+ if el is not None:
1259
+ href = el.get("href")
1260
+ if href:
1261
+ entry["link"] = href.strip()
1262
+
1263
+ # Dates: Atom 1.0 uses published/updated, Atom 0.3 uses issued/modified
1264
+ is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
1265
+ pub_tag = "issued" if is_atom_03 else "published"
1266
+ upd_tag = "modified" if is_atom_03 else "updated"
1267
+ pub_fallback_tag = "published" if is_atom_03 else "issued"
1268
+ upd_fallback_tag = "updated" if is_atom_03 else "modified"
1269
+
1270
+ el = item.find(ns + pub_tag)
1271
+ if el is not None and el.text:
1272
+ published = _parse_date(el.text)
1273
+ if published:
1274
+ entry["published"] = published
1275
+
1276
+ el = item.find(ns + upd_tag)
1277
+ if el is not None and el.text:
1278
+ updated = _parse_date(el.text)
1279
+ if updated:
1280
+ entry["updated"] = updated
1281
+
1282
+ # Fallback date fields for mixed namespace scenarios
1283
+ if "published" not in entry:
1284
+ el = item.find(ns + pub_fallback_tag)
1285
+ if el is not None and el.text:
1286
+ published = _parse_date(el.text)
1287
+ if published:
1288
+ entry["published"] = published
1289
+
1290
+ if "updated" not in entry:
1291
+ el = item.find(ns + upd_fallback_tag)
1292
+ if el is not None and el.text:
1293
+ updated = _parse_date(el.text)
1294
+ if updated:
1295
+ entry["updated"] = updated
1296
+
1297
+ if "updated" in entry and "published" not in entry:
1298
+ entry["published"] = entry["updated"]
1299
+
1300
+ _populate_entry_links(entry, item, atom_ns)
1301
+
1302
+ if "id" not in entry and "link" in entry:
1303
+ entry["id"] = entry["link"]
1304
+
1305
+ _populate_entry_content(entry, item, "atom", atom_ns)
1306
+
1307
+ if has_media_ns:
1308
+ media_contents = _parse_media_content(item)
1309
+ if media_contents:
1310
+ entry["media_content"] = media_contents
1311
+
1312
+ enclosures = _parse_enclosures(item)
1313
+ if enclosures:
1314
+ entry["enclosures"] = enclosures
1315
+
1316
+ # Author
1317
+ el = item.find(ns + "author/" + ns + "name")
1318
+ if el is not None and el.text:
1319
+ entry["author"] = el.text.strip()
1320
+
1321
+ tags = _parse_tags(item, "atom", atom_ns)
1322
+ if tags:
1323
+ entry["tags"] = tags
1324
+
1325
+ return entry
1326
+
1327
+
1196
1328
  def _parse_feed_entry(
1197
1329
  item: _Element,
1198
1330
  feed_type: _FeedType,
1199
1331
  atom_namespace: Optional[str] = None,
1332
+ has_media_ns: bool = True,
1200
1333
  ) -> FastFeedParserDict:
1201
1334
  # Use dynamic atom namespace or fallback to default
1202
1335
  atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
1203
1336
 
1204
1337
  if feed_type == "rss":
1205
- return _parse_rss_feed_entry_fast(item, atom_ns)
1338
+ return _parse_rss_feed_entry_fast(item, atom_ns, has_media_ns)
1206
1339
 
1340
+ if feed_type == "atom":
1341
+ return _parse_atom_feed_entry_fast(item, atom_ns, has_media_ns)
1342
+
1343
+ # RDF path uses the generic field machinery
1207
1344
  # Check if this is Atom 0.3 to use different date field names
1208
1345
  is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
1209
1346
 
@@ -1315,9 +1452,10 @@ def _parse_feed_entry(
1315
1452
 
1316
1453
  _populate_entry_content(entry, item, feed_type, atom_ns)
1317
1454
 
1318
- media_contents = _parse_media_content(item)
1319
- if media_contents:
1320
- entry["media_content"] = media_contents
1455
+ if has_media_ns:
1456
+ media_contents = _parse_media_content(item)
1457
+ if media_contents:
1458
+ entry["media_content"] = media_contents
1321
1459
 
1322
1460
  enclosures = _parse_enclosures(item)
1323
1461
  if enclosures:
@@ -1476,6 +1614,14 @@ def _normalize_iso_datetime_string(value: str) -> str:
1476
1614
  if not cleaned:
1477
1615
  return cleaned
1478
1616
 
1617
+ # Fast path: 'Z' suffix (most common in Atom feeds)
1618
+ if cleaned[-1] in ("Z", "z"):
1619
+ return cleaned[:-1] + "+00:00"
1620
+
1621
+ # Fast path: already has proper +HH:MM or -HH:MM timezone
1622
+ if len(cleaned) > 6 and cleaned[-6] in ("+", "-") and cleaned[-3] == ":":
1623
+ return cleaned
1624
+
1479
1625
  upper_cleaned = cleaned.upper()
1480
1626
  for suffix in (" UTC", " GMT", " Z"):
1481
1627
  if upper_cleaned.endswith(suffix):
@@ -1511,8 +1657,54 @@ def _ensure_utc(dt: datetime.datetime) -> Optional[datetime.datetime]:
1511
1657
  return None
1512
1658
 
1513
1659
 
1660
+ def _fast_rfc822_to_iso(value: str) -> Optional[str]:
1661
+ """Fast RFC-822 date to ISO string, bypassing datetime objects for UTC dates."""
1662
+ m = _RE_RFC822.match(value)
1663
+ if not m:
1664
+ return None
1665
+ day, mon_str, year, hour, minute, second, tz = m.groups()
1666
+ month = _MONTHS_RFC822.get(mon_str.lower())
1667
+ if month is None:
1668
+ return None
1669
+ if tz[0] in "+-":
1670
+ tz_offset_seconds = (int(tz[1:3]) * 3600 + int(tz[3:5]) * 60) * (
1671
+ 1 if tz[0] == "+" else -1
1672
+ )
1673
+ else:
1674
+ tz_offset_seconds = _TZ_OFFSETS_RFC822.get(tz)
1675
+ if tz_offset_seconds is None:
1676
+ return None # Unknown tz name, fall through to full parser
1677
+ # Python requires offset strictly between -24h and +24h
1678
+ if not (-86400 < tz_offset_seconds < 86400):
1679
+ return None
1680
+ d = int(day)
1681
+ h = int(hour)
1682
+ mi = int(minute)
1683
+ s = int(second)
1684
+ # Hour 24 is invalid (even ISO only allows 24:00:00); roll to next day at 00:mm:ss
1685
+ if h == 24:
1686
+ base = datetime.date(int(year), month, d) + datetime.timedelta(days=1)
1687
+ h = 0
1688
+ if tz_offset_seconds == 0:
1689
+ return f"{base.year:04d}-{base.month:02d}-{base.day:02d}T{h:02d}:{mi:02d}:{s:02d}+00:00"
1690
+ dt = datetime.datetime(
1691
+ base.year, base.month, base.day, h, mi, s,
1692
+ tzinfo=datetime.timezone(datetime.timedelta(seconds=tz_offset_seconds)),
1693
+ )
1694
+ utc = dt.astimezone(_UTC)
1695
+ return f"{utc.year:04d}-{utc.month:02d}-{utc.day:02d}T{utc.hour:02d}:{utc.minute:02d}:{utc.second:02d}+00:00"
1696
+ if tz_offset_seconds == 0:
1697
+ return f"{year}-{month:02d}-{d:02d}T{hour}:{minute}:{second}+00:00"
1698
+ dt = datetime.datetime(
1699
+ int(year), month, d, h, mi, s,
1700
+ tzinfo=datetime.timezone(datetime.timedelta(seconds=tz_offset_seconds)),
1701
+ )
1702
+ utc = dt.astimezone(_UTC)
1703
+ return f"{utc.year:04d}-{utc.month:02d}-{utc.day:02d}T{utc.hour:02d}:{utc.minute:02d}:{utc.second:02d}+00:00"
1704
+
1705
+
1514
1706
  def _parsedate_to_utc(value: str) -> Optional[datetime.datetime]:
1515
- """Fast RFC-822 / RFC-2822 parsing via email.utils."""
1707
+ """RFC-822 / RFC-2822 parsing via email.utils (fallback)."""
1516
1708
  try:
1517
1709
  parsed = parsedate_to_datetime(value)
1518
1710
  except (TypeError, ValueError, IndexError):
@@ -1585,7 +1777,7 @@ def _slow_dateparser(value: str) -> Optional[datetime.datetime]:
1585
1777
  except ImportError:
1586
1778
  return None
1587
1779
  try:
1588
- return _dateparser.parse(value, languages=["en"], settings=_DATEPARSER_SETTINGS)
1780
+ return _dateparser.parse(value, languages=["en"], settings={**_DATEPARSER_SETTINGS})
1589
1781
  except (ValueError, TypeError):
1590
1782
  return None
1591
1783
 
@@ -1605,6 +1797,30 @@ def _parse_date(date_str: str) -> Optional[str]:
1605
1797
  candidate = date_str.strip()
1606
1798
  if not candidate:
1607
1799
  return None
1800
+
1801
+ # Fast path: clean ISO-8601 (covers >90% of Atom/modern RSS dates)
1802
+ clen = len(candidate)
1803
+ if clen >= 20 and candidate[4] == "-" and candidate[0:4].isdigit():
1804
+ last = candidate[-1]
1805
+ # Most common: ends with 'Z' (e.g., 2024-01-15T10:30:00Z)
1806
+ if last in ("Z", "z"):
1807
+ iso = candidate[:-1] + "+00:00"
1808
+ try:
1809
+ dt = datetime.datetime.fromisoformat(iso)
1810
+ return dt.isoformat()
1811
+ except ValueError:
1812
+ pass # Fall through to full parsing
1813
+ # Second most common: ends with +HH:MM (e.g., 2024-01-15T10:30:00+00:00)
1814
+ elif clen > 6 and candidate[-6] in ("+", "-") and candidate[-3] == ":":
1815
+ try:
1816
+ dt = datetime.datetime.fromisoformat(candidate)
1817
+ if dt.tzinfo is _UTC:
1818
+ return dt.isoformat()
1819
+ utc_dt = dt.astimezone(_UTC)
1820
+ return utc_dt.isoformat()
1821
+ except (ValueError, OverflowError):
1822
+ pass # Fall through to full parsing
1823
+
1608
1824
  if "\n" in candidate or "\r" in candidate or "\t" in candidate or " " in candidate:
1609
1825
  candidate = _RE_WHITESPACE.sub(" ", candidate)
1610
1826
 
@@ -1617,10 +1833,13 @@ def _parse_date(date_str: str) -> Optional[str]:
1617
1833
  if not ((year % 4 == 0 and year % 100 != 0) or (year % 400 == 0)):
1618
1834
  candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
1619
1835
 
1620
- if "24:00" in candidate:
1621
- candidate = candidate.replace("24:00:00", "00:00:00").replace(
1622
- " 24:00", " 00:00"
1623
- )
1836
+ if "T24:" in candidate or " 24:" in candidate:
1837
+ m24 = re.search(r"(\d{4}-\d{2}-\d{2})[T ]24:(\d{2}):(\d{2})", candidate)
1838
+ if m24:
1839
+ base = datetime.date.fromisoformat(m24.group(1))
1840
+ mins, secs = int(m24.group(2)), int(m24.group(3))
1841
+ next_day = base + datetime.timedelta(days=1)
1842
+ candidate = candidate[:m24.start()] + f"{next_day}T00:{mins:02d}:{secs:02d}" + candidate[m24.end():]
1624
1843
 
1625
1844
  dt: Optional[datetime.datetime] = None
1626
1845
 
@@ -1636,6 +1855,10 @@ def _parse_date(date_str: str) -> Optional[str]:
1636
1855
  if utc_dt is not None:
1637
1856
  return utc_dt.isoformat()
1638
1857
 
1858
+ rfc822_result = _fast_rfc822_to_iso(candidate)
1859
+ if rfc822_result is not None:
1860
+ return rfc822_result
1861
+
1639
1862
  dt = _parsedate_to_utc(candidate)
1640
1863
  if dt is not None:
1641
1864
  return dt.isoformat()
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.1
3
+ Version: 0.5.3
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
File without changes
File without changes