fastfeedparser 0.5.4__tar.gz → 0.5.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fastfeedparser-0.5.4/src/fastfeedparser.egg-info → fastfeedparser-0.5.5}/PKG-INFO +2 -2
- {fastfeedparser-0.5.4 → fastfeedparser-0.5.5}/README.md +1 -1
- {fastfeedparser-0.5.4 → fastfeedparser-0.5.5}/setup.cfg +1 -1
- {fastfeedparser-0.5.4 → fastfeedparser-0.5.5}/src/fastfeedparser/main.py +62 -41
- {fastfeedparser-0.5.4 → fastfeedparser-0.5.5/src/fastfeedparser.egg-info}/PKG-INFO +2 -2
- {fastfeedparser-0.5.4 → fastfeedparser-0.5.5}/LICENSE +0 -0
- {fastfeedparser-0.5.4 → fastfeedparser-0.5.5}/pyproject.toml +0 -0
- {fastfeedparser-0.5.4 → fastfeedparser-0.5.5}/src/fastfeedparser/__init__.py +0 -0
- {fastfeedparser-0.5.4 → fastfeedparser-0.5.5}/src/fastfeedparser.egg-info/SOURCES.txt +0 -0
- {fastfeedparser-0.5.4 → fastfeedparser-0.5.5}/src/fastfeedparser.egg-info/dependency_links.txt +0 -0
- {fastfeedparser-0.5.4 → fastfeedparser-0.5.5}/src/fastfeedparser.egg-info/requires.txt +0 -0
- {fastfeedparser-0.5.4 → fastfeedparser-0.5.5}/src/fastfeedparser.egg-info/top_level.txt +0 -0
- {fastfeedparser-0.5.4 → fastfeedparser-0.5.5}/tests/test_encoding.py +0 -0
- {fastfeedparser-0.5.4 → fastfeedparser-0.5.5}/tests/test_integration.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: fastfeedparser
|
|
3
|
-
Version: 0.5.
|
|
3
|
+
Version: 0.5.5
|
|
4
4
|
Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
|
|
5
5
|
Home-page: https://github.com/kagisearch/fastfeedparser
|
|
6
6
|
Author: Vladimir Prelovac
|
|
@@ -34,7 +34,7 @@ A high-performance feed parser for Python that handles RSS, Atom, and RDF. Built
|
|
|
34
34
|
|
|
35
35
|
### Why FastFeedParser?
|
|
36
36
|
|
|
37
|
-
It's about
|
|
37
|
+
It's about 25x faster (check included `benchmark.py`) than popular feedparser
|
|
38
38
|
library while keeping a familiar API. This speed comes from:
|
|
39
39
|
|
|
40
40
|
- lxml for efficient XML parsing
|
|
@@ -4,7 +4,7 @@ A high-performance feed parser for Python that handles RSS, Atom, and RDF. Built
|
|
|
4
4
|
|
|
5
5
|
### Why FastFeedParser?
|
|
6
6
|
|
|
7
|
-
It's about
|
|
7
|
+
It's about 25x faster (check included `benchmark.py`) than popular feedparser
|
|
8
8
|
library while keeping a familiar API. This speed comes from:
|
|
9
9
|
|
|
10
10
|
- lxml for efficient XML parsing
|
|
@@ -15,7 +15,7 @@ try:
|
|
|
15
15
|
HAS_BROTLI = True
|
|
16
16
|
except ImportError:
|
|
17
17
|
HAS_BROTLI = False
|
|
18
|
-
from typing import Any, Callable, Optional,
|
|
18
|
+
from typing import Any, Callable, Optional, TYPE_CHECKING, Literal
|
|
19
19
|
from urllib.parse import urljoin
|
|
20
20
|
from urllib.request import (
|
|
21
21
|
HTTPErrorProcessor,
|
|
@@ -28,13 +28,14 @@ from dateutil import parser as dateutil_parser
|
|
|
28
28
|
from lxml import etree
|
|
29
29
|
|
|
30
30
|
if TYPE_CHECKING:
|
|
31
|
-
from
|
|
31
|
+
from typing import Protocol
|
|
32
32
|
|
|
33
|
-
|
|
33
|
+
from lxml.etree import _Element
|
|
34
34
|
|
|
35
|
+
class _ElementValueGetter(Protocol):
|
|
36
|
+
def __call__(self, path: str, attribute: Optional[str] = None) -> Optional[str]: ...
|
|
35
37
|
|
|
36
|
-
|
|
37
|
-
def __call__(self, path: str, attribute: Optional[str] = None) -> Optional[str]: ...
|
|
38
|
+
_FeedType = Literal["rss", "atom", "rdf"]
|
|
38
39
|
|
|
39
40
|
|
|
40
41
|
_UTC = datetime.timezone.utc
|
|
@@ -64,6 +65,7 @@ _RE_ISO_FRACTION = re.compile(r"\.(\d{7,})(?=(?:[+-]\d{2}:?\d{2}|Z|$))", re.IGNO
|
|
|
64
65
|
_RE_RFC822 = re.compile(
|
|
65
66
|
r"(?:\w{3},\s+)?(\d{1,2})\s+(\w{3})\s+(\d{4})\s+(\d{2}):(\d{2}):(\d{2})\s+([+-]\d{4}|[A-Z]{2,5})"
|
|
66
67
|
)
|
|
68
|
+
_RE_HOUR24 = re.compile(r"(\d{4}-\d{2}-\d{2})[T ]24:(\d{2}):(\d{2})")
|
|
67
69
|
_MONTHS_RFC822: dict[str, int] = {
|
|
68
70
|
"jan": 1,
|
|
69
71
|
"feb": 2,
|
|
@@ -78,19 +80,6 @@ _MONTHS_RFC822: dict[str, int] = {
|
|
|
78
80
|
"nov": 11,
|
|
79
81
|
"dec": 12,
|
|
80
82
|
}
|
|
81
|
-
_TZ_OFFSETS_RFC822: dict[str, int] = {
|
|
82
|
-
"GMT": 0,
|
|
83
|
-
"UTC": 0,
|
|
84
|
-
"UT": 0,
|
|
85
|
-
"EST": -18000,
|
|
86
|
-
"EDT": -14400,
|
|
87
|
-
"CST": -21600,
|
|
88
|
-
"CDT": -18000,
|
|
89
|
-
"MST": -25200,
|
|
90
|
-
"MDT": -21600,
|
|
91
|
-
"PST": -28800,
|
|
92
|
-
"PDT": -25200,
|
|
93
|
-
}
|
|
94
83
|
|
|
95
84
|
|
|
96
85
|
class FastFeedParserDict(dict):
|
|
@@ -206,6 +195,13 @@ def _prepare_xml_bytes(xml_content: str | bytes) -> bytes:
|
|
|
206
195
|
if not cleaned.strip():
|
|
207
196
|
raise ValueError("Empty content")
|
|
208
197
|
|
|
198
|
+
# Replace Unicode LINE SEPARATOR (U+2028) and PARAGRAPH SEPARATOR (U+2029)
|
|
199
|
+
# with regular newlines — these are invalid in XML 1.0 and cause lxml to fail.
|
|
200
|
+
if b"\xe2\x80\xa8" in cleaned or b"\xe2\x80\xa9" in cleaned:
|
|
201
|
+
cleaned = cleaned.replace(b"\xe2\x80\xa8", b"\n").replace(
|
|
202
|
+
b"\xe2\x80\xa9", b"\n"
|
|
203
|
+
)
|
|
204
|
+
|
|
209
205
|
detected_encoding = _detect_xml_encoding(cleaned)
|
|
210
206
|
actual_encoding = detected_encoding
|
|
211
207
|
if detected_encoding.startswith("utf-16") and b"\x00" not in cleaned[:200]:
|
|
@@ -695,20 +691,28 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
|
695
691
|
else "search.yahoo.com/mrss" in xml_content
|
|
696
692
|
)
|
|
697
693
|
|
|
698
|
-
# Parse entries
|
|
694
|
+
# Parse entries — resolve parser once per feed instead of per entry
|
|
699
695
|
entries: list[FastFeedParserDict] = []
|
|
700
696
|
feed["entries"] = entries
|
|
701
|
-
|
|
702
|
-
|
|
703
|
-
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
710
|
-
|
|
711
|
-
|
|
697
|
+
atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
|
|
698
|
+
if feed_type == "rss":
|
|
699
|
+
for item in items:
|
|
700
|
+
entry = _parse_rss_feed_entry_fast(item, atom_ns, has_media_ns)
|
|
701
|
+
entry.setdefault("title", "")
|
|
702
|
+
entry.setdefault("description", "")
|
|
703
|
+
entries.append(entry)
|
|
704
|
+
elif feed_type == "atom":
|
|
705
|
+
for item in items:
|
|
706
|
+
entry = _parse_atom_feed_entry_fast(item, atom_ns, has_media_ns)
|
|
707
|
+
entry.setdefault("title", "")
|
|
708
|
+
entry.setdefault("description", "")
|
|
709
|
+
entries.append(entry)
|
|
710
|
+
else:
|
|
711
|
+
for item in items:
|
|
712
|
+
entry = _parse_feed_entry(item, feed_type, atom_namespace, has_media_ns)
|
|
713
|
+
entry["title"] = entry.get("title", "").strip()
|
|
714
|
+
entry["description"] = entry.get("description", "").strip()
|
|
715
|
+
entries.append(entry)
|
|
712
716
|
|
|
713
717
|
return feed
|
|
714
718
|
|
|
@@ -1015,13 +1019,30 @@ def _populate_entry_links(
|
|
|
1015
1019
|
|
|
1016
1020
|
|
|
1017
1021
|
def _populate_entry_content(
|
|
1018
|
-
entry: FastFeedParserDict,
|
|
1022
|
+
entry: FastFeedParserDict,
|
|
1023
|
+
item: _Element,
|
|
1024
|
+
feed_type: _FeedType,
|
|
1025
|
+
atom_ns: str,
|
|
1026
|
+
rss_text_by_full: Optional[dict[str, Optional[str]]] = None,
|
|
1019
1027
|
) -> None:
|
|
1020
1028
|
content_el = None
|
|
1021
1029
|
if feed_type == "rss":
|
|
1022
|
-
|
|
1023
|
-
if
|
|
1024
|
-
|
|
1030
|
+
# Fast path: check pre-built text map before doing tree searches
|
|
1031
|
+
if rss_text_by_full is not None:
|
|
1032
|
+
content_encoded_text = rss_text_by_full.get(
|
|
1033
|
+
"{http://purl.org/rss/1.0/modules/content/}encoded"
|
|
1034
|
+
)
|
|
1035
|
+
if content_encoded_text is not None:
|
|
1036
|
+
# We have content:encoded text — still need the element for type/lang/base attrs
|
|
1037
|
+
content_el = item.find("{http://purl.org/rss/1.0/modules/content/}encoded")
|
|
1038
|
+
else:
|
|
1039
|
+
content_text = rss_text_by_full.get("content")
|
|
1040
|
+
if content_text is not None:
|
|
1041
|
+
content_el = item.find("content")
|
|
1042
|
+
else:
|
|
1043
|
+
content_el = item.find("{http://purl.org/rss/1.0/modules/content/}encoded")
|
|
1044
|
+
if content_el is None:
|
|
1045
|
+
content_el = item.find("content")
|
|
1025
1046
|
elif feed_type == "atom":
|
|
1026
1047
|
content_el = item.find(f"{{{atom_ns}}}content")
|
|
1027
1048
|
|
|
@@ -1058,7 +1079,7 @@ def _populate_entry_content(
|
|
|
1058
1079
|
content_value = entry["content"][0]["value"]
|
|
1059
1080
|
if content_value:
|
|
1060
1081
|
if "<" in content_value:
|
|
1061
|
-
content_value = _RE_HTML_TAGS.sub(" ", content_value[:
|
|
1082
|
+
content_value = _RE_HTML_TAGS.sub(" ", content_value[:1024])
|
|
1062
1083
|
content_value = _html_mod.unescape(content_value)
|
|
1063
1084
|
if (
|
|
1064
1085
|
" " in content_value
|
|
@@ -1210,11 +1231,11 @@ def _parse_rss_feed_entry_fast(
|
|
|
1210
1231
|
|
|
1211
1232
|
title = text_by_local.get("title")
|
|
1212
1233
|
if title:
|
|
1213
|
-
entry["title"] = title
|
|
1234
|
+
entry["title"] = title.strip()
|
|
1214
1235
|
|
|
1215
1236
|
description = _first_non_empty(text_by_local, ("description", "summary"))
|
|
1216
1237
|
if description:
|
|
1217
|
-
entry["description"] = description
|
|
1238
|
+
entry["description"] = description.strip()
|
|
1218
1239
|
|
|
1219
1240
|
link = text_by_local.get("link")
|
|
1220
1241
|
if link:
|
|
@@ -1266,7 +1287,7 @@ def _parse_rss_feed_entry_fast(
|
|
|
1266
1287
|
if "id" not in entry and "link" in entry:
|
|
1267
1288
|
entry["id"] = entry["link"]
|
|
1268
1289
|
|
|
1269
|
-
_populate_entry_content(entry, item, "rss", atom_ns)
|
|
1290
|
+
_populate_entry_content(entry, item, "rss", atom_ns, rss_text_by_full=text_by_full)
|
|
1270
1291
|
|
|
1271
1292
|
if has_media_ns:
|
|
1272
1293
|
media_contents = _parse_media_content(item)
|
|
@@ -1758,7 +1779,7 @@ def _fast_rfc822_to_iso(value: str) -> Optional[str]:
|
|
|
1758
1779
|
1 if tz[0] == "+" else -1
|
|
1759
1780
|
)
|
|
1760
1781
|
else:
|
|
1761
|
-
tz_offset_seconds =
|
|
1782
|
+
tz_offset_seconds = _custom_tzinfos.get(tz)
|
|
1762
1783
|
if tz_offset_seconds is None:
|
|
1763
1784
|
return None # Unknown tz name, fall through to full parser
|
|
1764
1785
|
# Python requires offset strictly between -24h and +24h
|
|
@@ -1875,7 +1896,7 @@ def _slow_dateparser(value: str) -> Optional[datetime.datetime]:
|
|
|
1875
1896
|
return None
|
|
1876
1897
|
try:
|
|
1877
1898
|
return _dateparser.parse(
|
|
1878
|
-
value, languages=["en"], settings=
|
|
1899
|
+
value, languages=["en"], settings=_DATEPARSER_SETTINGS
|
|
1879
1900
|
)
|
|
1880
1901
|
except (ValueError, TypeError):
|
|
1881
1902
|
return None
|
|
@@ -1933,7 +1954,7 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1933
1954
|
candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
|
|
1934
1955
|
|
|
1935
1956
|
if "T24:" in candidate or " 24:" in candidate:
|
|
1936
|
-
m24 =
|
|
1957
|
+
m24 = _RE_HOUR24.search(candidate)
|
|
1937
1958
|
if m24:
|
|
1938
1959
|
base = datetime.date.fromisoformat(m24.group(1))
|
|
1939
1960
|
mins, secs = int(m24.group(2)), int(m24.group(3))
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: fastfeedparser
|
|
3
|
-
Version: 0.5.
|
|
3
|
+
Version: 0.5.5
|
|
4
4
|
Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
|
|
5
5
|
Home-page: https://github.com/kagisearch/fastfeedparser
|
|
6
6
|
Author: Vladimir Prelovac
|
|
@@ -34,7 +34,7 @@ A high-performance feed parser for Python that handles RSS, Atom, and RDF. Built
|
|
|
34
34
|
|
|
35
35
|
### Why FastFeedParser?
|
|
36
36
|
|
|
37
|
-
It's about
|
|
37
|
+
It's about 25x faster (check included `benchmark.py`) than popular feedparser
|
|
38
38
|
library while keeping a familiar API. This speed comes from:
|
|
39
39
|
|
|
40
40
|
- lxml for efficient XML parsing
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{fastfeedparser-0.5.4 → fastfeedparser-0.5.5}/src/fastfeedparser.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|