fastfeedparser 0.5.4__tar.gz → 0.5.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.4
3
+ Version: 0.5.5
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -34,7 +34,7 @@ A high-performance feed parser for Python that handles RSS, Atom, and RDF. Built
34
34
 
35
35
  ### Why FastFeedParser?
36
36
 
37
- It's about 10x faster (check included `benchmark.py`) than popular feedparser
37
+ It's about 25x faster (check included `benchmark.py`) than popular feedparser
38
38
  library while keeping a familiar API. This speed comes from:
39
39
 
40
40
  - lxml for efficient XML parsing
@@ -4,7 +4,7 @@ A high-performance feed parser for Python that handles RSS, Atom, and RDF. Built
4
4
 
5
5
  ### Why FastFeedParser?
6
6
 
7
- It's about 10x faster (check included `benchmark.py`) than popular feedparser
7
+ It's about 25x faster (check included `benchmark.py`) than popular feedparser
8
8
  library while keeping a familiar API. This speed comes from:
9
9
 
10
10
  - lxml for efficient XML parsing
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = fastfeedparser
3
- version = 0.5.4
3
+ version = 0.5.5
4
4
  author = Vladimir Prelovac
5
5
  author_email = vlad@kagi.com
6
6
  description = High performance RSS, Atom, JSON and RDF feed parser in Python
@@ -15,7 +15,7 @@ try:
15
15
  HAS_BROTLI = True
16
16
  except ImportError:
17
17
  HAS_BROTLI = False
18
- from typing import Any, Callable, Optional, Protocol, TYPE_CHECKING, Literal
18
+ from typing import Any, Callable, Optional, TYPE_CHECKING, Literal
19
19
  from urllib.parse import urljoin
20
20
  from urllib.request import (
21
21
  HTTPErrorProcessor,
@@ -28,13 +28,14 @@ from dateutil import parser as dateutil_parser
28
28
  from lxml import etree
29
29
 
30
30
  if TYPE_CHECKING:
31
- from lxml.etree import _Element
31
+ from typing import Protocol
32
32
 
33
- _FeedType = Literal["rss", "atom", "rdf"]
33
+ from lxml.etree import _Element
34
34
 
35
+ class _ElementValueGetter(Protocol):
36
+ def __call__(self, path: str, attribute: Optional[str] = None) -> Optional[str]: ...
35
37
 
36
- class _ElementValueGetter(Protocol):
37
- def __call__(self, path: str, attribute: Optional[str] = None) -> Optional[str]: ...
38
+ _FeedType = Literal["rss", "atom", "rdf"]
38
39
 
39
40
 
40
41
  _UTC = datetime.timezone.utc
@@ -64,6 +65,7 @@ _RE_ISO_FRACTION = re.compile(r"\.(\d{7,})(?=(?:[+-]\d{2}:?\d{2}|Z|$))", re.IGNO
64
65
  _RE_RFC822 = re.compile(
65
66
  r"(?:\w{3},\s+)?(\d{1,2})\s+(\w{3})\s+(\d{4})\s+(\d{2}):(\d{2}):(\d{2})\s+([+-]\d{4}|[A-Z]{2,5})"
66
67
  )
68
+ _RE_HOUR24 = re.compile(r"(\d{4}-\d{2}-\d{2})[T ]24:(\d{2}):(\d{2})")
67
69
  _MONTHS_RFC822: dict[str, int] = {
68
70
  "jan": 1,
69
71
  "feb": 2,
@@ -78,19 +80,6 @@ _MONTHS_RFC822: dict[str, int] = {
78
80
  "nov": 11,
79
81
  "dec": 12,
80
82
  }
81
- _TZ_OFFSETS_RFC822: dict[str, int] = {
82
- "GMT": 0,
83
- "UTC": 0,
84
- "UT": 0,
85
- "EST": -18000,
86
- "EDT": -14400,
87
- "CST": -21600,
88
- "CDT": -18000,
89
- "MST": -25200,
90
- "MDT": -21600,
91
- "PST": -28800,
92
- "PDT": -25200,
93
- }
94
83
 
95
84
 
96
85
  class FastFeedParserDict(dict):
@@ -206,6 +195,13 @@ def _prepare_xml_bytes(xml_content: str | bytes) -> bytes:
206
195
  if not cleaned.strip():
207
196
  raise ValueError("Empty content")
208
197
 
198
+ # Replace Unicode LINE SEPARATOR (U+2028) and PARAGRAPH SEPARATOR (U+2029)
199
+ # with regular newlines — these are invalid in XML 1.0 and cause lxml to fail.
200
+ if b"\xe2\x80\xa8" in cleaned or b"\xe2\x80\xa9" in cleaned:
201
+ cleaned = cleaned.replace(b"\xe2\x80\xa8", b"\n").replace(
202
+ b"\xe2\x80\xa9", b"\n"
203
+ )
204
+
209
205
  detected_encoding = _detect_xml_encoding(cleaned)
210
206
  actual_encoding = detected_encoding
211
207
  if detected_encoding.startswith("utf-16") and b"\x00" not in cleaned[:200]:
@@ -695,20 +691,28 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
695
691
  else "search.yahoo.com/mrss" in xml_content
696
692
  )
697
693
 
698
- # Parse entries
694
+ # Parse entries — resolve parser once per feed instead of per entry
699
695
  entries: list[FastFeedParserDict] = []
700
696
  feed["entries"] = entries
701
- for item in items:
702
- entry = _parse_feed_entry(
703
- item,
704
- feed_type,
705
- atom_namespace,
706
- has_media_ns,
707
- )
708
- # Ensure that titles and descriptions are always present
709
- entry["title"] = entry.get("title", "").strip()
710
- entry["description"] = entry.get("description", "").strip()
711
- entries.append(entry)
697
+ atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
698
+ if feed_type == "rss":
699
+ for item in items:
700
+ entry = _parse_rss_feed_entry_fast(item, atom_ns, has_media_ns)
701
+ entry.setdefault("title", "")
702
+ entry.setdefault("description", "")
703
+ entries.append(entry)
704
+ elif feed_type == "atom":
705
+ for item in items:
706
+ entry = _parse_atom_feed_entry_fast(item, atom_ns, has_media_ns)
707
+ entry.setdefault("title", "")
708
+ entry.setdefault("description", "")
709
+ entries.append(entry)
710
+ else:
711
+ for item in items:
712
+ entry = _parse_feed_entry(item, feed_type, atom_namespace, has_media_ns)
713
+ entry["title"] = entry.get("title", "").strip()
714
+ entry["description"] = entry.get("description", "").strip()
715
+ entries.append(entry)
712
716
 
713
717
  return feed
714
718
 
@@ -1015,13 +1019,30 @@ def _populate_entry_links(
1015
1019
 
1016
1020
 
1017
1021
  def _populate_entry_content(
1018
- entry: FastFeedParserDict, item: _Element, feed_type: _FeedType, atom_ns: str
1022
+ entry: FastFeedParserDict,
1023
+ item: _Element,
1024
+ feed_type: _FeedType,
1025
+ atom_ns: str,
1026
+ rss_text_by_full: Optional[dict[str, Optional[str]]] = None,
1019
1027
  ) -> None:
1020
1028
  content_el = None
1021
1029
  if feed_type == "rss":
1022
- content_el = item.find("{http://purl.org/rss/1.0/modules/content/}encoded")
1023
- if content_el is None:
1024
- content_el = item.find("content")
1030
+ # Fast path: check pre-built text map before doing tree searches
1031
+ if rss_text_by_full is not None:
1032
+ content_encoded_text = rss_text_by_full.get(
1033
+ "{http://purl.org/rss/1.0/modules/content/}encoded"
1034
+ )
1035
+ if content_encoded_text is not None:
1036
+ # We have content:encoded text — still need the element for type/lang/base attrs
1037
+ content_el = item.find("{http://purl.org/rss/1.0/modules/content/}encoded")
1038
+ else:
1039
+ content_text = rss_text_by_full.get("content")
1040
+ if content_text is not None:
1041
+ content_el = item.find("content")
1042
+ else:
1043
+ content_el = item.find("{http://purl.org/rss/1.0/modules/content/}encoded")
1044
+ if content_el is None:
1045
+ content_el = item.find("content")
1025
1046
  elif feed_type == "atom":
1026
1047
  content_el = item.find(f"{{{atom_ns}}}content")
1027
1048
 
@@ -1058,7 +1079,7 @@ def _populate_entry_content(
1058
1079
  content_value = entry["content"][0]["value"]
1059
1080
  if content_value:
1060
1081
  if "<" in content_value:
1061
- content_value = _RE_HTML_TAGS.sub(" ", content_value[:2048])
1082
+ content_value = _RE_HTML_TAGS.sub(" ", content_value[:1024])
1062
1083
  content_value = _html_mod.unescape(content_value)
1063
1084
  if (
1064
1085
  " " in content_value
@@ -1210,11 +1231,11 @@ def _parse_rss_feed_entry_fast(
1210
1231
 
1211
1232
  title = text_by_local.get("title")
1212
1233
  if title:
1213
- entry["title"] = title
1234
+ entry["title"] = title.strip()
1214
1235
 
1215
1236
  description = _first_non_empty(text_by_local, ("description", "summary"))
1216
1237
  if description:
1217
- entry["description"] = description
1238
+ entry["description"] = description.strip()
1218
1239
 
1219
1240
  link = text_by_local.get("link")
1220
1241
  if link:
@@ -1266,7 +1287,7 @@ def _parse_rss_feed_entry_fast(
1266
1287
  if "id" not in entry and "link" in entry:
1267
1288
  entry["id"] = entry["link"]
1268
1289
 
1269
- _populate_entry_content(entry, item, "rss", atom_ns)
1290
+ _populate_entry_content(entry, item, "rss", atom_ns, rss_text_by_full=text_by_full)
1270
1291
 
1271
1292
  if has_media_ns:
1272
1293
  media_contents = _parse_media_content(item)
@@ -1758,7 +1779,7 @@ def _fast_rfc822_to_iso(value: str) -> Optional[str]:
1758
1779
  1 if tz[0] == "+" else -1
1759
1780
  )
1760
1781
  else:
1761
- tz_offset_seconds = _TZ_OFFSETS_RFC822.get(tz)
1782
+ tz_offset_seconds = _custom_tzinfos.get(tz)
1762
1783
  if tz_offset_seconds is None:
1763
1784
  return None # Unknown tz name, fall through to full parser
1764
1785
  # Python requires offset strictly between -24h and +24h
@@ -1875,7 +1896,7 @@ def _slow_dateparser(value: str) -> Optional[datetime.datetime]:
1875
1896
  return None
1876
1897
  try:
1877
1898
  return _dateparser.parse(
1878
- value, languages=["en"], settings={**_DATEPARSER_SETTINGS}
1899
+ value, languages=["en"], settings=_DATEPARSER_SETTINGS
1879
1900
  )
1880
1901
  except (ValueError, TypeError):
1881
1902
  return None
@@ -1933,7 +1954,7 @@ def _parse_date(date_str: str) -> Optional[str]:
1933
1954
  candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
1934
1955
 
1935
1956
  if "T24:" in candidate or " 24:" in candidate:
1936
- m24 = re.search(r"(\d{4}-\d{2}-\d{2})[T ]24:(\d{2}):(\d{2})", candidate)
1957
+ m24 = _RE_HOUR24.search(candidate)
1937
1958
  if m24:
1938
1959
  base = datetime.date.fromisoformat(m24.group(1))
1939
1960
  mins, secs = int(m24.group(2)), int(m24.group(3))
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.4
3
+ Version: 0.5.5
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -34,7 +34,7 @@ A high-performance feed parser for Python that handles RSS, Atom, and RDF. Built
34
34
 
35
35
  ### Why FastFeedParser?
36
36
 
37
- It's about 10x faster (check included `benchmark.py`) than popular feedparser
37
+ It's about 25x faster (check included `benchmark.py`) than popular feedparser
38
38
  library while keeping a familiar API. This speed comes from:
39
39
 
40
40
  - lxml for efficient XML parsing
File without changes