fastfeedparser 0.5.3__tar.gz → 0.5.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fastfeedparser-0.5.3/src/fastfeedparser.egg-info → fastfeedparser-0.5.5}/PKG-INFO +2 -2
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.5}/README.md +1 -1
- fastfeedparser-0.5.5/pyproject.toml +23 -0
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.5}/setup.cfg +1 -1
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.5}/src/fastfeedparser/main.py +197 -71
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.5/src/fastfeedparser.egg-info}/PKG-INFO +2 -2
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.5}/tests/test_encoding.py +8 -3
- fastfeedparser-0.5.3/pyproject.toml +0 -7
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.5}/LICENSE +0 -0
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.5}/src/fastfeedparser/__init__.py +0 -0
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.5}/src/fastfeedparser.egg-info/SOURCES.txt +0 -0
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.5}/src/fastfeedparser.egg-info/dependency_links.txt +0 -0
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.5}/src/fastfeedparser.egg-info/requires.txt +0 -0
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.5}/src/fastfeedparser.egg-info/top_level.txt +0 -0
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.5}/tests/test_integration.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: fastfeedparser
|
|
3
|
-
Version: 0.5.
|
|
3
|
+
Version: 0.5.5
|
|
4
4
|
Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
|
|
5
5
|
Home-page: https://github.com/kagisearch/fastfeedparser
|
|
6
6
|
Author: Vladimir Prelovac
|
|
@@ -34,7 +34,7 @@ A high-performance feed parser for Python that handles RSS, Atom, and RDF. Built
|
|
|
34
34
|
|
|
35
35
|
### Why FastFeedParser?
|
|
36
36
|
|
|
37
|
-
It's about
|
|
37
|
+
It's about 25x faster (check included `benchmark.py`) than popular feedparser
|
|
38
38
|
library while keeping a familiar API. This speed comes from:
|
|
39
39
|
|
|
40
40
|
- lxml for efficient XML parsing
|
|
@@ -4,7 +4,7 @@ A high-performance feed parser for Python that handles RSS, Atom, and RDF. Built
|
|
|
4
4
|
|
|
5
5
|
### Why FastFeedParser?
|
|
6
6
|
|
|
7
|
-
It's about
|
|
7
|
+
It's about 25x faster (check included `benchmark.py`) than popular feedparser
|
|
8
8
|
library while keeping a familiar API. This speed comes from:
|
|
9
9
|
|
|
10
10
|
- lxml for efficient XML parsing
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools~=67.0", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[tool.ruff]
|
|
6
|
+
extend-exclude = [
|
|
7
|
+
"1.py",
|
|
8
|
+
"debug_*.py",
|
|
9
|
+
"test_*.py",
|
|
10
|
+
"benchmark.py",
|
|
11
|
+
"investigate_failures.py",
|
|
12
|
+
"show_error_messages.py",
|
|
13
|
+
"check_oh4_dates.py",
|
|
14
|
+
"comprehensive_debug.py",
|
|
15
|
+
"profile_dylanharris.py",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
[tool.ty.rules]
|
|
19
|
+
unresolved-import = "ignore"
|
|
20
|
+
|
|
21
|
+
[tool.pytest.ini_options]
|
|
22
|
+
testpaths = ["tests"]
|
|
23
|
+
|
|
@@ -28,10 +28,16 @@ from dateutil import parser as dateutil_parser
|
|
|
28
28
|
from lxml import etree
|
|
29
29
|
|
|
30
30
|
if TYPE_CHECKING:
|
|
31
|
+
from typing import Protocol
|
|
32
|
+
|
|
31
33
|
from lxml.etree import _Element
|
|
32
34
|
|
|
35
|
+
class _ElementValueGetter(Protocol):
|
|
36
|
+
def __call__(self, path: str, attribute: Optional[str] = None) -> Optional[str]: ...
|
|
37
|
+
|
|
33
38
|
_FeedType = Literal["rss", "atom", "rdf"]
|
|
34
39
|
|
|
40
|
+
|
|
35
41
|
_UTC = datetime.timezone.utc
|
|
36
42
|
|
|
37
43
|
# Pre-compiled regex patterns for performance
|
|
@@ -39,16 +45,16 @@ _RE_XML_DECL_ENCODING = re.compile(
|
|
|
39
45
|
r'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
|
|
40
46
|
)
|
|
41
47
|
_RE_XML_DECL_ENCODING_BYTES = re.compile(
|
|
42
|
-
|
|
48
|
+
rb'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
|
|
43
49
|
)
|
|
44
|
-
_RE_DOUBLE_XML_DECL_BYTES = re.compile(
|
|
45
|
-
_RE_DOUBLE_CLOSE_BYTES = re.compile(
|
|
46
|
-
_RE_UNQUOTED_ATTR_BYTES = re.compile(
|
|
50
|
+
_RE_DOUBLE_XML_DECL_BYTES = re.compile(rb"<\?xml\?xml\s+", re.IGNORECASE)
|
|
51
|
+
_RE_DOUBLE_CLOSE_BYTES = re.compile(rb"\?\?>\s*")
|
|
52
|
+
_RE_UNQUOTED_ATTR_BYTES = re.compile(rb'(\s+[\w:]+)=([^\s>"\']+)')
|
|
47
53
|
_RE_UTF16_ENCODING_BYTES = re.compile(
|
|
48
|
-
|
|
54
|
+
rb'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
|
|
49
55
|
)
|
|
50
56
|
_RE_UNCLOSED_LINK_BYTES = re.compile(
|
|
51
|
-
|
|
57
|
+
rb"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
|
|
52
58
|
)
|
|
53
59
|
_RE_FEB29 = re.compile(r"(\d{4})-02-29")
|
|
54
60
|
_RE_HTML_TAGS = re.compile(r"<[^>]+>")
|
|
@@ -59,14 +65,20 @@ _RE_ISO_FRACTION = re.compile(r"\.(\d{7,})(?=(?:[+-]\d{2}:?\d{2}|Z|$))", re.IGNO
|
|
|
59
65
|
_RE_RFC822 = re.compile(
|
|
60
66
|
r"(?:\w{3},\s+)?(\d{1,2})\s+(\w{3})\s+(\d{4})\s+(\d{2}):(\d{2}):(\d{2})\s+([+-]\d{4}|[A-Z]{2,5})"
|
|
61
67
|
)
|
|
68
|
+
_RE_HOUR24 = re.compile(r"(\d{4}-\d{2}-\d{2})[T ]24:(\d{2}):(\d{2})")
|
|
62
69
|
_MONTHS_RFC822: dict[str, int] = {
|
|
63
|
-
"jan": 1,
|
|
64
|
-
"
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
"
|
|
68
|
-
"
|
|
69
|
-
"
|
|
70
|
+
"jan": 1,
|
|
71
|
+
"feb": 2,
|
|
72
|
+
"mar": 3,
|
|
73
|
+
"apr": 4,
|
|
74
|
+
"may": 5,
|
|
75
|
+
"jun": 6,
|
|
76
|
+
"jul": 7,
|
|
77
|
+
"aug": 8,
|
|
78
|
+
"sep": 9,
|
|
79
|
+
"oct": 10,
|
|
80
|
+
"nov": 11,
|
|
81
|
+
"dec": 12,
|
|
70
82
|
}
|
|
71
83
|
|
|
72
84
|
|
|
@@ -129,7 +141,9 @@ def _clean_feed_bytes(content: bytes) -> bytes:
|
|
|
129
141
|
if preview_lower.startswith((b"<?xml", b"<rss", b"<feed", b"<rdf")):
|
|
130
142
|
return stripped_content
|
|
131
143
|
|
|
132
|
-
if preview_lower.startswith(b"<!doctype html") or preview_lower.startswith(
|
|
144
|
+
if preview_lower.startswith(b"<!doctype html") or preview_lower.startswith(
|
|
145
|
+
b"<html"
|
|
146
|
+
):
|
|
133
147
|
raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
|
|
134
148
|
|
|
135
149
|
xml_start_patterns = (
|
|
@@ -160,15 +174,17 @@ def _fix_malformed_xml_bytes(content: bytes, actual_encoding: str = "utf-8") ->
|
|
|
160
174
|
content = _RE_DOUBLE_CLOSE_BYTES.sub(b"?>", content)
|
|
161
175
|
|
|
162
176
|
# Fix malformed attribute syntax like rss:version=2.0 (missing quotes)
|
|
163
|
-
content = _RE_UNQUOTED_ATTR_BYTES.sub(
|
|
177
|
+
content = _RE_UNQUOTED_ATTR_BYTES.sub(rb'\1="\2"', content)
|
|
164
178
|
|
|
165
179
|
# Update encoding in XML declaration to match actual encoding when a feed was transcoded.
|
|
166
180
|
if actual_encoding.lower() != "utf-16":
|
|
167
|
-
replacement =
|
|
181
|
+
replacement = (
|
|
182
|
+
rb"\1" + actual_encoding.encode("ascii", errors="replace") + rb"\3"
|
|
183
|
+
)
|
|
168
184
|
content = _RE_UTF16_ENCODING_BYTES.sub(replacement, content)
|
|
169
185
|
|
|
170
186
|
# Fix unclosed link tags - common in Atom feeds
|
|
171
|
-
content = _RE_UNCLOSED_LINK_BYTES.sub(
|
|
187
|
+
content = _RE_UNCLOSED_LINK_BYTES.sub(rb"<link\1/>", content)
|
|
172
188
|
|
|
173
189
|
return content
|
|
174
190
|
|
|
@@ -179,6 +195,13 @@ def _prepare_xml_bytes(xml_content: str | bytes) -> bytes:
|
|
|
179
195
|
if not cleaned.strip():
|
|
180
196
|
raise ValueError("Empty content")
|
|
181
197
|
|
|
198
|
+
# Replace Unicode LINE SEPARATOR (U+2028) and PARAGRAPH SEPARATOR (U+2029)
|
|
199
|
+
# with regular newlines — these are invalid in XML 1.0 and cause lxml to fail.
|
|
200
|
+
if b"\xe2\x80\xa8" in cleaned or b"\xe2\x80\xa9" in cleaned:
|
|
201
|
+
cleaned = cleaned.replace(b"\xe2\x80\xa8", b"\n").replace(
|
|
202
|
+
b"\xe2\x80\xa9", b"\n"
|
|
203
|
+
)
|
|
204
|
+
|
|
182
205
|
detected_encoding = _detect_xml_encoding(cleaned)
|
|
183
206
|
actual_encoding = detected_encoding
|
|
184
207
|
if detected_encoding.startswith("utf-16") and b"\x00" not in cleaned[:200]:
|
|
@@ -500,16 +523,16 @@ def _raise_for_non_feed_root(
|
|
|
500
523
|
if base_msg is None:
|
|
501
524
|
return
|
|
502
525
|
|
|
503
|
-
error_msg =
|
|
526
|
+
error_msg = (
|
|
527
|
+
_extract_error_message(root, raw_bytes).strip()[:300] or "No error message"
|
|
528
|
+
)
|
|
504
529
|
|
|
505
530
|
if error_msg != "No error message" and len(error_msg) > 10:
|
|
506
531
|
raise ValueError(f"{base_msg}: {error_msg[:150]}")
|
|
507
532
|
raise ValueError(base_msg)
|
|
508
533
|
|
|
509
534
|
|
|
510
|
-
_RE_META_REFRESH_URL = re.compile(
|
|
511
|
-
r'url\s*=\s*["\']?\s*([^"\'>\s]+)', re.IGNORECASE
|
|
512
|
-
)
|
|
535
|
+
_RE_META_REFRESH_URL = re.compile(r'url\s*=\s*["\']?\s*([^"\'>\s]+)', re.IGNORECASE)
|
|
513
536
|
|
|
514
537
|
|
|
515
538
|
def _extract_meta_refresh_url(content: str | bytes, base_url: str) -> str | None:
|
|
@@ -558,7 +581,8 @@ def _detect_feed_structure(
|
|
|
558
581
|
if channel is None:
|
|
559
582
|
has_atom_elements = any(
|
|
560
583
|
isinstance(child.tag, str)
|
|
561
|
-
and child.tag
|
|
584
|
+
and child.tag
|
|
585
|
+
in {"entry", "title", "subtitle", "updated", "id", "author", "link"}
|
|
562
586
|
for child in root
|
|
563
587
|
)
|
|
564
588
|
if has_atom_elements:
|
|
@@ -586,7 +610,9 @@ def _detect_feed_structure(
|
|
|
586
610
|
items = []
|
|
587
611
|
items.append(child)
|
|
588
612
|
if not items:
|
|
589
|
-
items = channel.xpath(".//item") or channel.xpath(
|
|
613
|
+
items = channel.xpath(".//item") or channel.xpath(
|
|
614
|
+
".//*[local-name()='item']"
|
|
615
|
+
)
|
|
590
616
|
|
|
591
617
|
if not items:
|
|
592
618
|
items = channel.findall("entry")
|
|
@@ -633,7 +659,9 @@ def _detect_feed_structure(
|
|
|
633
659
|
if root.tag == "{http://www.w3.org/1999/02/22-rdf-syntax-ns#}RDF":
|
|
634
660
|
feed_type = "rdf"
|
|
635
661
|
channel = root
|
|
636
|
-
items = channel.findall(".//{http://purl.org/rss/1.0/}item") or channel.findall(
|
|
662
|
+
items = channel.findall(".//{http://purl.org/rss/1.0/}item") or channel.findall(
|
|
663
|
+
"item"
|
|
664
|
+
)
|
|
637
665
|
return feed_type, channel, items, atom_namespace
|
|
638
666
|
|
|
639
667
|
raise ValueError(f"Unknown feed type: {root.tag}")
|
|
@@ -657,22 +685,34 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
|
657
685
|
feed = _parse_feed_info(channel, feed_type, atom_namespace)
|
|
658
686
|
|
|
659
687
|
# Detect once whether media namespace is used anywhere in the document
|
|
660
|
-
has_media_ns =
|
|
688
|
+
has_media_ns = (
|
|
689
|
+
b"search.yahoo.com/mrss" in xml_content
|
|
690
|
+
if isinstance(xml_content, bytes)
|
|
691
|
+
else "search.yahoo.com/mrss" in xml_content
|
|
692
|
+
)
|
|
661
693
|
|
|
662
|
-
# Parse entries
|
|
694
|
+
# Parse entries — resolve parser once per feed instead of per entry
|
|
663
695
|
entries: list[FastFeedParserDict] = []
|
|
664
696
|
feed["entries"] = entries
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
|
|
668
|
-
|
|
669
|
-
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
|
|
697
|
+
atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
|
|
698
|
+
if feed_type == "rss":
|
|
699
|
+
for item in items:
|
|
700
|
+
entry = _parse_rss_feed_entry_fast(item, atom_ns, has_media_ns)
|
|
701
|
+
entry.setdefault("title", "")
|
|
702
|
+
entry.setdefault("description", "")
|
|
703
|
+
entries.append(entry)
|
|
704
|
+
elif feed_type == "atom":
|
|
705
|
+
for item in items:
|
|
706
|
+
entry = _parse_atom_feed_entry_fast(item, atom_ns, has_media_ns)
|
|
707
|
+
entry.setdefault("title", "")
|
|
708
|
+
entry.setdefault("description", "")
|
|
709
|
+
entries.append(entry)
|
|
710
|
+
else:
|
|
711
|
+
for item in items:
|
|
712
|
+
entry = _parse_feed_entry(item, feed_type, atom_namespace, has_media_ns)
|
|
713
|
+
entry["title"] = entry.get("title", "").strip()
|
|
714
|
+
entry["description"] = entry.get("description", "").strip()
|
|
715
|
+
entries.append(entry)
|
|
676
716
|
|
|
677
717
|
return feed
|
|
678
718
|
|
|
@@ -692,6 +732,7 @@ def parse(source: str | bytes) -> FastFeedParserDict:
|
|
|
692
732
|
"""
|
|
693
733
|
is_url = isinstance(source, str) and source.startswith(("http://", "https://"))
|
|
694
734
|
if is_url:
|
|
735
|
+
assert isinstance(source, str)
|
|
695
736
|
content = _fetch_url_content(source)
|
|
696
737
|
else:
|
|
697
738
|
content = source
|
|
@@ -701,6 +742,7 @@ def parse(source: str | bytes) -> FastFeedParserDict:
|
|
|
701
742
|
except ValueError as e:
|
|
702
743
|
if not is_url:
|
|
703
744
|
raise
|
|
745
|
+
assert isinstance(source, str)
|
|
704
746
|
err_msg = str(e)
|
|
705
747
|
if "HTML" not in err_msg and "not a valid RSS/Atom feed" not in err_msg:
|
|
706
748
|
raise
|
|
@@ -930,7 +972,9 @@ def _coerce_int_fields(mapping: dict[str, Any], fields: tuple[str, ...]) -> None
|
|
|
930
972
|
mapping.pop(field, None)
|
|
931
973
|
|
|
932
974
|
|
|
933
|
-
def _populate_entry_links(
|
|
975
|
+
def _populate_entry_links(
|
|
976
|
+
entry: FastFeedParserDict, item: _Element, atom_ns: str
|
|
977
|
+
) -> None:
|
|
934
978
|
entry_links: list[dict[str, Optional[str]]] = []
|
|
935
979
|
alternate_link: Optional[dict[str, Optional[str]]] = None
|
|
936
980
|
for link in item.findall(f"{{{atom_ns}}}link"):
|
|
@@ -951,7 +995,9 @@ def _populate_entry_links(entry: FastFeedParserDict, item: _Element, atom_ns: st
|
|
|
951
995
|
|
|
952
996
|
guid = item.find("guid")
|
|
953
997
|
guid_text = guid.text.strip() if guid is not None and guid.text else None
|
|
954
|
-
is_guid_url = guid_text is not None and guid_text.startswith(
|
|
998
|
+
is_guid_url = guid_text is not None and guid_text.startswith(
|
|
999
|
+
("http://", "https://")
|
|
1000
|
+
)
|
|
955
1001
|
|
|
956
1002
|
if is_guid_url and "link" not in entry:
|
|
957
1003
|
entry["link"] = guid_text
|
|
@@ -973,13 +1019,30 @@ def _populate_entry_links(entry: FastFeedParserDict, item: _Element, atom_ns: st
|
|
|
973
1019
|
|
|
974
1020
|
|
|
975
1021
|
def _populate_entry_content(
|
|
976
|
-
entry: FastFeedParserDict,
|
|
1022
|
+
entry: FastFeedParserDict,
|
|
1023
|
+
item: _Element,
|
|
1024
|
+
feed_type: _FeedType,
|
|
1025
|
+
atom_ns: str,
|
|
1026
|
+
rss_text_by_full: Optional[dict[str, Optional[str]]] = None,
|
|
977
1027
|
) -> None:
|
|
978
1028
|
content_el = None
|
|
979
1029
|
if feed_type == "rss":
|
|
980
|
-
|
|
981
|
-
if
|
|
982
|
-
|
|
1030
|
+
# Fast path: check pre-built text map before doing tree searches
|
|
1031
|
+
if rss_text_by_full is not None:
|
|
1032
|
+
content_encoded_text = rss_text_by_full.get(
|
|
1033
|
+
"{http://purl.org/rss/1.0/modules/content/}encoded"
|
|
1034
|
+
)
|
|
1035
|
+
if content_encoded_text is not None:
|
|
1036
|
+
# We have content:encoded text — still need the element for type/lang/base attrs
|
|
1037
|
+
content_el = item.find("{http://purl.org/rss/1.0/modules/content/}encoded")
|
|
1038
|
+
else:
|
|
1039
|
+
content_text = rss_text_by_full.get("content")
|
|
1040
|
+
if content_text is not None:
|
|
1041
|
+
content_el = item.find("content")
|
|
1042
|
+
else:
|
|
1043
|
+
content_el = item.find("{http://purl.org/rss/1.0/modules/content/}encoded")
|
|
1044
|
+
if content_el is None:
|
|
1045
|
+
content_el = item.find("content")
|
|
983
1046
|
elif feed_type == "atom":
|
|
984
1047
|
content_el = item.find(f"{{{atom_ns}}}content")
|
|
985
1048
|
|
|
@@ -992,7 +1055,9 @@ def _populate_entry_content(
|
|
|
992
1055
|
entry["content"] = [
|
|
993
1056
|
{
|
|
994
1057
|
"type": content_type,
|
|
995
|
-
"language": content_el.get(
|
|
1058
|
+
"language": content_el.get(
|
|
1059
|
+
"{http://www.w3.org/XML/1998/namespace}lang"
|
|
1060
|
+
),
|
|
996
1061
|
"base": content_el.get("{http://www.w3.org/XML/1998/namespace}base"),
|
|
997
1062
|
"value": content_value,
|
|
998
1063
|
}
|
|
@@ -1014,9 +1079,14 @@ def _populate_entry_content(
|
|
|
1014
1079
|
content_value = entry["content"][0]["value"]
|
|
1015
1080
|
if content_value:
|
|
1016
1081
|
if "<" in content_value:
|
|
1017
|
-
content_value = _RE_HTML_TAGS.sub(" ", content_value[:
|
|
1082
|
+
content_value = _RE_HTML_TAGS.sub(" ", content_value[:1024])
|
|
1018
1083
|
content_value = _html_mod.unescape(content_value)
|
|
1019
|
-
if
|
|
1084
|
+
if (
|
|
1085
|
+
" " in content_value
|
|
1086
|
+
or "\n" in content_value
|
|
1087
|
+
or "\t" in content_value
|
|
1088
|
+
or "\r" in content_value
|
|
1089
|
+
):
|
|
1020
1090
|
content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
|
|
1021
1091
|
else:
|
|
1022
1092
|
content_value = content_value.strip()
|
|
@@ -1110,7 +1180,9 @@ def _parse_enclosures(item: _Element) -> list[dict[str, Any]] | None:
|
|
|
1110
1180
|
return enclosures or None
|
|
1111
1181
|
|
|
1112
1182
|
|
|
1113
|
-
def _build_rss_item_text_maps(
|
|
1183
|
+
def _build_rss_item_text_maps(
|
|
1184
|
+
item: _Element,
|
|
1185
|
+
) -> tuple[dict[str, Optional[str]], dict[str, Optional[str]]]:
|
|
1114
1186
|
by_local: dict[str, Optional[str]] = {}
|
|
1115
1187
|
by_full: dict[str, Optional[str]] = {}
|
|
1116
1188
|
for child in item:
|
|
@@ -1132,7 +1204,9 @@ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]],
|
|
|
1132
1204
|
return by_local, by_full
|
|
1133
1205
|
|
|
1134
1206
|
|
|
1135
|
-
def _first_non_empty(
|
|
1207
|
+
def _first_non_empty(
|
|
1208
|
+
mapping: dict[str, Optional[str]], keys: tuple[str, ...]
|
|
1209
|
+
) -> Optional[str]:
|
|
1136
1210
|
for key in keys:
|
|
1137
1211
|
value = mapping.get(key)
|
|
1138
1212
|
if value:
|
|
@@ -1157,29 +1231,37 @@ def _parse_rss_feed_entry_fast(
|
|
|
1157
1231
|
|
|
1158
1232
|
title = text_by_local.get("title")
|
|
1159
1233
|
if title:
|
|
1160
|
-
entry["title"] = title
|
|
1234
|
+
entry["title"] = title.strip()
|
|
1161
1235
|
|
|
1162
1236
|
description = _first_non_empty(text_by_local, ("description", "summary"))
|
|
1163
1237
|
if description:
|
|
1164
|
-
entry["description"] = description
|
|
1238
|
+
entry["description"] = description.strip()
|
|
1165
1239
|
|
|
1166
1240
|
link = text_by_local.get("link")
|
|
1167
1241
|
if link:
|
|
1168
1242
|
entry["link"] = link.strip()
|
|
1169
1243
|
|
|
1170
|
-
published_source = _first_non_empty(
|
|
1244
|
+
published_source = _first_non_empty(
|
|
1245
|
+
text_by_local, ("pubdate", "published", "issued", "date")
|
|
1246
|
+
)
|
|
1171
1247
|
if published_source:
|
|
1172
1248
|
published = _parse_date(published_source)
|
|
1173
1249
|
if published:
|
|
1174
1250
|
entry["published"] = published
|
|
1175
1251
|
|
|
1176
|
-
updated_source = _first_non_empty(
|
|
1252
|
+
updated_source = _first_non_empty(
|
|
1253
|
+
text_by_local, ("lastbuilddate", "updated", "modified")
|
|
1254
|
+
)
|
|
1177
1255
|
if updated_source:
|
|
1178
1256
|
updated = _parse_date(updated_source)
|
|
1179
1257
|
if updated:
|
|
1180
1258
|
entry["updated"] = updated
|
|
1181
1259
|
|
|
1182
|
-
if
|
|
1260
|
+
if (
|
|
1261
|
+
"published" not in entry
|
|
1262
|
+
and rss_guid
|
|
1263
|
+
and not rss_guid.startswith(("http://", "https://"))
|
|
1264
|
+
):
|
|
1183
1265
|
guid_date = _parse_date(rss_guid)
|
|
1184
1266
|
if guid_date:
|
|
1185
1267
|
entry["published"] = guid_date
|
|
@@ -1195,13 +1277,17 @@ def _parse_rss_feed_entry_fast(
|
|
|
1195
1277
|
else:
|
|
1196
1278
|
# Common RSS case: no atom:link elements
|
|
1197
1279
|
entry["links"] = []
|
|
1198
|
-
if
|
|
1280
|
+
if (
|
|
1281
|
+
"link" not in entry
|
|
1282
|
+
and rss_guid
|
|
1283
|
+
and rss_guid.startswith(("http://", "https://"))
|
|
1284
|
+
):
|
|
1199
1285
|
entry["link"] = rss_guid
|
|
1200
1286
|
|
|
1201
1287
|
if "id" not in entry and "link" in entry:
|
|
1202
1288
|
entry["id"] = entry["link"]
|
|
1203
1289
|
|
|
1204
|
-
_populate_entry_content(entry, item, "rss", atom_ns)
|
|
1290
|
+
_populate_entry_content(entry, item, "rss", atom_ns, rss_text_by_full=text_by_full)
|
|
1205
1291
|
|
|
1206
1292
|
if has_media_ns:
|
|
1207
1293
|
media_contents = _parse_media_content(item)
|
|
@@ -1215,7 +1301,11 @@ def _parse_rss_feed_entry_fast(
|
|
|
1215
1301
|
author = _first_non_empty(text_by_local, ("author", "creator"))
|
|
1216
1302
|
if not author:
|
|
1217
1303
|
atom_author = item.find(f"{{{atom_ns}}}author/{{{atom_ns}}}name")
|
|
1218
|
-
author =
|
|
1304
|
+
author = (
|
|
1305
|
+
atom_author.text.strip()
|
|
1306
|
+
if atom_author is not None and atom_author.text
|
|
1307
|
+
else None
|
|
1308
|
+
)
|
|
1219
1309
|
if author:
|
|
1220
1310
|
entry["author"] = author.strip()
|
|
1221
1311
|
|
|
@@ -1432,7 +1522,11 @@ def _parse_feed_entry(
|
|
|
1432
1522
|
entry["updated"] = _parse_date(fallback_updated)
|
|
1433
1523
|
|
|
1434
1524
|
# Try to extract date from GUID as final fallback
|
|
1435
|
-
if
|
|
1525
|
+
if (
|
|
1526
|
+
"published" not in entry
|
|
1527
|
+
and rss_guid
|
|
1528
|
+
and not rss_guid.startswith(("http://", "https://"))
|
|
1529
|
+
):
|
|
1436
1530
|
guid_date = _parse_date(rss_guid)
|
|
1437
1531
|
if guid_date:
|
|
1438
1532
|
entry["published"] = guid_date
|
|
@@ -1468,7 +1562,9 @@ def _parse_feed_entry(
|
|
|
1468
1562
|
False,
|
|
1469
1563
|
)
|
|
1470
1564
|
if not author:
|
|
1471
|
-
author = element_get(
|
|
1565
|
+
author = element_get(
|
|
1566
|
+
"{http://purl.org/dc/elements/1.1/}creator"
|
|
1567
|
+
) or element_get("author")
|
|
1472
1568
|
if author:
|
|
1473
1569
|
entry["author"] = author
|
|
1474
1570
|
|
|
@@ -1483,9 +1579,9 @@ def _parse_feed_entry(
|
|
|
1483
1579
|
def _field_value_getter(
|
|
1484
1580
|
root: _Element,
|
|
1485
1581
|
feed_type: _FeedType,
|
|
1486
|
-
cached_get: Optional[
|
|
1582
|
+
cached_get: Optional[_ElementValueGetter] = None,
|
|
1487
1583
|
) -> Callable[[str, str, str, bool], str | None]:
|
|
1488
|
-
get_value = cached_get or _cached_element_value_factory(root)
|
|
1584
|
+
get_value: _ElementValueGetter = cached_get or _cached_element_value_factory(root)
|
|
1489
1585
|
|
|
1490
1586
|
if feed_type == "rss":
|
|
1491
1587
|
|
|
@@ -1574,7 +1670,11 @@ def _get_element_value(
|
|
|
1574
1670
|
el = found
|
|
1575
1671
|
break
|
|
1576
1672
|
else:
|
|
1577
|
-
prefixed_paths = [
|
|
1673
|
+
prefixed_paths = [
|
|
1674
|
+
f"rss:{path_lower}",
|
|
1675
|
+
f"atom:{path_lower}",
|
|
1676
|
+
f"dc:{path_lower}",
|
|
1677
|
+
]
|
|
1578
1678
|
for child in root:
|
|
1579
1679
|
if not isinstance(child.tag, str):
|
|
1580
1680
|
continue
|
|
@@ -1594,7 +1694,7 @@ def _get_element_value(
|
|
|
1594
1694
|
|
|
1595
1695
|
def _cached_element_value_factory(
|
|
1596
1696
|
root: _Element,
|
|
1597
|
-
) ->
|
|
1697
|
+
) -> _ElementValueGetter:
|
|
1598
1698
|
"""Create a closure with a child tag index for fast namespace-prefix lookups."""
|
|
1599
1699
|
# Build child tag index once: O(children) instead of O(children × misses)
|
|
1600
1700
|
child_index: dict[str, _Element] = {}
|
|
@@ -1603,7 +1703,9 @@ def _cached_element_value_factory(
|
|
|
1603
1703
|
child_index[child.tag.lower()] = child
|
|
1604
1704
|
|
|
1605
1705
|
def getter(path: str, attribute: Optional[str] = None) -> Optional[str]:
|
|
1606
|
-
return _get_element_value(
|
|
1706
|
+
return _get_element_value(
|
|
1707
|
+
root, path, attribute=attribute, child_index=child_index
|
|
1708
|
+
)
|
|
1607
1709
|
|
|
1608
1710
|
return getter
|
|
1609
1711
|
|
|
@@ -1632,7 +1734,13 @@ def _normalize_iso_datetime_string(value: str) -> str:
|
|
|
1632
1734
|
if cleaned.endswith(("Z", "z")):
|
|
1633
1735
|
cleaned = cleaned[:-1] + "+00:00"
|
|
1634
1736
|
|
|
1635
|
-
if
|
|
1737
|
+
if (
|
|
1738
|
+
" " in cleaned
|
|
1739
|
+
and "T" not in cleaned[:11]
|
|
1740
|
+
and len(cleaned) >= 10
|
|
1741
|
+
and cleaned[4] == "-"
|
|
1742
|
+
and cleaned[0:4].isdigit()
|
|
1743
|
+
):
|
|
1636
1744
|
date_part, rest = cleaned.split(" ", 1)
|
|
1637
1745
|
if rest and rest[0].isdigit():
|
|
1638
1746
|
cleaned = f"{date_part}T{rest}"
|
|
@@ -1671,7 +1779,7 @@ def _fast_rfc822_to_iso(value: str) -> Optional[str]:
|
|
|
1671
1779
|
1 if tz[0] == "+" else -1
|
|
1672
1780
|
)
|
|
1673
1781
|
else:
|
|
1674
|
-
tz_offset_seconds =
|
|
1782
|
+
tz_offset_seconds = _custom_tzinfos.get(tz)
|
|
1675
1783
|
if tz_offset_seconds is None:
|
|
1676
1784
|
return None # Unknown tz name, fall through to full parser
|
|
1677
1785
|
# Python requires offset strictly between -24h and +24h
|
|
@@ -1688,7 +1796,12 @@ def _fast_rfc822_to_iso(value: str) -> Optional[str]:
|
|
|
1688
1796
|
if tz_offset_seconds == 0:
|
|
1689
1797
|
return f"{base.year:04d}-{base.month:02d}-{base.day:02d}T{h:02d}:{mi:02d}:{s:02d}+00:00"
|
|
1690
1798
|
dt = datetime.datetime(
|
|
1691
|
-
base.year,
|
|
1799
|
+
base.year,
|
|
1800
|
+
base.month,
|
|
1801
|
+
base.day,
|
|
1802
|
+
h,
|
|
1803
|
+
mi,
|
|
1804
|
+
s,
|
|
1692
1805
|
tzinfo=datetime.timezone(datetime.timedelta(seconds=tz_offset_seconds)),
|
|
1693
1806
|
)
|
|
1694
1807
|
utc = dt.astimezone(_UTC)
|
|
@@ -1696,7 +1809,12 @@ def _fast_rfc822_to_iso(value: str) -> Optional[str]:
|
|
|
1696
1809
|
if tz_offset_seconds == 0:
|
|
1697
1810
|
return f"{year}-{month:02d}-{d:02d}T{hour}:{minute}:{second}+00:00"
|
|
1698
1811
|
dt = datetime.datetime(
|
|
1699
|
-
int(year),
|
|
1812
|
+
int(year),
|
|
1813
|
+
month,
|
|
1814
|
+
d,
|
|
1815
|
+
h,
|
|
1816
|
+
mi,
|
|
1817
|
+
s,
|
|
1700
1818
|
tzinfo=datetime.timezone(datetime.timedelta(seconds=tz_offset_seconds)),
|
|
1701
1819
|
)
|
|
1702
1820
|
utc = dt.astimezone(_UTC)
|
|
@@ -1777,7 +1895,9 @@ def _slow_dateparser(value: str) -> Optional[datetime.datetime]:
|
|
|
1777
1895
|
except ImportError:
|
|
1778
1896
|
return None
|
|
1779
1897
|
try:
|
|
1780
|
-
return _dateparser.parse(
|
|
1898
|
+
return _dateparser.parse(
|
|
1899
|
+
value, languages=["en"], settings=_DATEPARSER_SETTINGS
|
|
1900
|
+
)
|
|
1781
1901
|
except (ValueError, TypeError):
|
|
1782
1902
|
return None
|
|
1783
1903
|
|
|
@@ -1834,16 +1954,22 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1834
1954
|
candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
|
|
1835
1955
|
|
|
1836
1956
|
if "T24:" in candidate or " 24:" in candidate:
|
|
1837
|
-
m24 =
|
|
1957
|
+
m24 = _RE_HOUR24.search(candidate)
|
|
1838
1958
|
if m24:
|
|
1839
1959
|
base = datetime.date.fromisoformat(m24.group(1))
|
|
1840
1960
|
mins, secs = int(m24.group(2)), int(m24.group(3))
|
|
1841
1961
|
next_day = base + datetime.timedelta(days=1)
|
|
1842
|
-
candidate =
|
|
1962
|
+
candidate = (
|
|
1963
|
+
candidate[: m24.start()]
|
|
1964
|
+
+ f"{next_day}T00:{mins:02d}:{secs:02d}"
|
|
1965
|
+
+ candidate[m24.end() :]
|
|
1966
|
+
)
|
|
1843
1967
|
|
|
1844
1968
|
dt: Optional[datetime.datetime] = None
|
|
1845
1969
|
|
|
1846
|
-
is_iso_like =
|
|
1970
|
+
is_iso_like = (
|
|
1971
|
+
len(candidate) >= 10 and candidate[4] == "-" and candidate[0:4].isdigit()
|
|
1972
|
+
)
|
|
1847
1973
|
if is_iso_like:
|
|
1848
1974
|
iso_candidate = _normalize_iso_datetime_string(candidate)
|
|
1849
1975
|
try:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: fastfeedparser
|
|
3
|
-
Version: 0.5.
|
|
3
|
+
Version: 0.5.5
|
|
4
4
|
Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
|
|
5
5
|
Home-page: https://github.com/kagisearch/fastfeedparser
|
|
6
6
|
Author: Vladimir Prelovac
|
|
@@ -34,7 +34,7 @@ A high-performance feed parser for Python that handles RSS, Atom, and RDF. Built
|
|
|
34
34
|
|
|
35
35
|
### Why FastFeedParser?
|
|
36
36
|
|
|
37
|
-
It's about
|
|
37
|
+
It's about 25x faster (check included `benchmark.py`) than popular feedparser
|
|
38
38
|
library while keeping a familiar API. This speed comes from:
|
|
39
39
|
|
|
40
40
|
- lxml for efficient XML parsing
|
|
@@ -34,12 +34,18 @@ def test_parse_bytes_with_non_utf8_encoding():
|
|
|
34
34
|
|
|
35
35
|
def test_meta_refresh_extraction():
|
|
36
36
|
html = '<!doctype html><html><head><meta http-equiv=refresh content="0; url=https://example.com/feed.xml"></head></html>'
|
|
37
|
-
assert
|
|
37
|
+
assert (
|
|
38
|
+
_extract_meta_refresh_url(html, "https://example.com/feed/")
|
|
39
|
+
== "https://example.com/feed.xml"
|
|
40
|
+
)
|
|
38
41
|
|
|
39
42
|
|
|
40
43
|
def test_meta_refresh_relative_url():
|
|
41
44
|
html = b'<html><head><meta http-equiv="refresh" content="0;url=/index.xml"></head></html>'
|
|
42
|
-
assert
|
|
45
|
+
assert (
|
|
46
|
+
_extract_meta_refresh_url(html, "https://example.com/feed/")
|
|
47
|
+
== "https://example.com/index.xml"
|
|
48
|
+
)
|
|
43
49
|
|
|
44
50
|
|
|
45
51
|
def test_meta_refresh_none_when_missing():
|
|
@@ -50,4 +56,3 @@ def test_meta_refresh_none_when_missing():
|
|
|
50
56
|
def test_meta_refresh_none_when_same_url():
|
|
51
57
|
html = '<html><head><meta http-equiv="refresh" content="0; url=https://example.com/"></head></html>'
|
|
52
58
|
assert _extract_meta_refresh_url(html, "https://example.com/") is None
|
|
53
|
-
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{fastfeedparser-0.5.3 → fastfeedparser-0.5.5}/src/fastfeedparser.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|