fastfeedparser 0.5.2__tar.gz → 0.5.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fastfeedparser-0.5.2/src/fastfeedparser.egg-info → fastfeedparser-0.5.4}/PKG-INFO +1 -1
- fastfeedparser-0.5.4/pyproject.toml +23 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.4}/setup.cfg +1 -1
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.4}/src/fastfeedparser/main.py +276 -74
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.4/src/fastfeedparser.egg-info}/PKG-INFO +1 -1
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.4}/tests/test_encoding.py +8 -3
- fastfeedparser-0.5.2/pyproject.toml +0 -7
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.4}/LICENSE +0 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.4}/README.md +0 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.4}/src/fastfeedparser/__init__.py +0 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.4}/src/fastfeedparser.egg-info/SOURCES.txt +0 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.4}/src/fastfeedparser.egg-info/dependency_links.txt +0 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.4}/src/fastfeedparser.egg-info/requires.txt +0 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.4}/src/fastfeedparser.egg-info/top_level.txt +0 -0
- {fastfeedparser-0.5.2 → fastfeedparser-0.5.4}/tests/test_integration.py +0 -0
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools~=67.0", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[tool.ruff]
|
|
6
|
+
extend-exclude = [
|
|
7
|
+
"1.py",
|
|
8
|
+
"debug_*.py",
|
|
9
|
+
"test_*.py",
|
|
10
|
+
"benchmark.py",
|
|
11
|
+
"investigate_failures.py",
|
|
12
|
+
"show_error_messages.py",
|
|
13
|
+
"check_oh4_dates.py",
|
|
14
|
+
"comprehensive_debug.py",
|
|
15
|
+
"profile_dylanharris.py",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
[tool.ty.rules]
|
|
19
|
+
unresolved-import = "ignore"
|
|
20
|
+
|
|
21
|
+
[tool.pytest.ini_options]
|
|
22
|
+
testpaths = ["tests"]
|
|
23
|
+
|
|
@@ -15,7 +15,7 @@ try:
|
|
|
15
15
|
HAS_BROTLI = True
|
|
16
16
|
except ImportError:
|
|
17
17
|
HAS_BROTLI = False
|
|
18
|
-
from typing import Any, Callable, Optional, TYPE_CHECKING, Literal
|
|
18
|
+
from typing import Any, Callable, Optional, Protocol, TYPE_CHECKING, Literal
|
|
19
19
|
from urllib.parse import urljoin
|
|
20
20
|
from urllib.request import (
|
|
21
21
|
HTTPErrorProcessor,
|
|
@@ -32,6 +32,11 @@ if TYPE_CHECKING:
|
|
|
32
32
|
|
|
33
33
|
_FeedType = Literal["rss", "atom", "rdf"]
|
|
34
34
|
|
|
35
|
+
|
|
36
|
+
class _ElementValueGetter(Protocol):
|
|
37
|
+
def __call__(self, path: str, attribute: Optional[str] = None) -> Optional[str]: ...
|
|
38
|
+
|
|
39
|
+
|
|
35
40
|
_UTC = datetime.timezone.utc
|
|
36
41
|
|
|
37
42
|
# Pre-compiled regex patterns for performance
|
|
@@ -39,16 +44,16 @@ _RE_XML_DECL_ENCODING = re.compile(
|
|
|
39
44
|
r'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
|
|
40
45
|
)
|
|
41
46
|
_RE_XML_DECL_ENCODING_BYTES = re.compile(
|
|
42
|
-
|
|
47
|
+
rb'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
|
|
43
48
|
)
|
|
44
|
-
_RE_DOUBLE_XML_DECL_BYTES = re.compile(
|
|
45
|
-
_RE_DOUBLE_CLOSE_BYTES = re.compile(
|
|
46
|
-
_RE_UNQUOTED_ATTR_BYTES = re.compile(
|
|
49
|
+
_RE_DOUBLE_XML_DECL_BYTES = re.compile(rb"<\?xml\?xml\s+", re.IGNORECASE)
|
|
50
|
+
_RE_DOUBLE_CLOSE_BYTES = re.compile(rb"\?\?>\s*")
|
|
51
|
+
_RE_UNQUOTED_ATTR_BYTES = re.compile(rb'(\s+[\w:]+)=([^\s>"\']+)')
|
|
47
52
|
_RE_UTF16_ENCODING_BYTES = re.compile(
|
|
48
|
-
|
|
53
|
+
rb'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
|
|
49
54
|
)
|
|
50
55
|
_RE_UNCLOSED_LINK_BYTES = re.compile(
|
|
51
|
-
|
|
56
|
+
rb"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
|
|
52
57
|
)
|
|
53
58
|
_RE_FEB29 = re.compile(r"(\d{4})-02-29")
|
|
54
59
|
_RE_HTML_TAGS = re.compile(r"<[^>]+>")
|
|
@@ -56,6 +61,36 @@ _RE_WHITESPACE = re.compile(r"\s+")
|
|
|
56
61
|
_RE_ISO_TZ_NO_COLON = re.compile(r"([+-]\d{2})(\d{2})$")
|
|
57
62
|
_RE_ISO_TZ_HOUR_ONLY = re.compile(r"([+-]\d{2})$")
|
|
58
63
|
_RE_ISO_FRACTION = re.compile(r"\.(\d{7,})(?=(?:[+-]\d{2}:?\d{2}|Z|$))", re.IGNORECASE)
|
|
64
|
+
_RE_RFC822 = re.compile(
|
|
65
|
+
r"(?:\w{3},\s+)?(\d{1,2})\s+(\w{3})\s+(\d{4})\s+(\d{2}):(\d{2}):(\d{2})\s+([+-]\d{4}|[A-Z]{2,5})"
|
|
66
|
+
)
|
|
67
|
+
_MONTHS_RFC822: dict[str, int] = {
|
|
68
|
+
"jan": 1,
|
|
69
|
+
"feb": 2,
|
|
70
|
+
"mar": 3,
|
|
71
|
+
"apr": 4,
|
|
72
|
+
"may": 5,
|
|
73
|
+
"jun": 6,
|
|
74
|
+
"jul": 7,
|
|
75
|
+
"aug": 8,
|
|
76
|
+
"sep": 9,
|
|
77
|
+
"oct": 10,
|
|
78
|
+
"nov": 11,
|
|
79
|
+
"dec": 12,
|
|
80
|
+
}
|
|
81
|
+
_TZ_OFFSETS_RFC822: dict[str, int] = {
|
|
82
|
+
"GMT": 0,
|
|
83
|
+
"UTC": 0,
|
|
84
|
+
"UT": 0,
|
|
85
|
+
"EST": -18000,
|
|
86
|
+
"EDT": -14400,
|
|
87
|
+
"CST": -21600,
|
|
88
|
+
"CDT": -18000,
|
|
89
|
+
"MST": -25200,
|
|
90
|
+
"MDT": -21600,
|
|
91
|
+
"PST": -28800,
|
|
92
|
+
"PDT": -25200,
|
|
93
|
+
}
|
|
59
94
|
|
|
60
95
|
|
|
61
96
|
class FastFeedParserDict(dict):
|
|
@@ -117,7 +152,9 @@ def _clean_feed_bytes(content: bytes) -> bytes:
|
|
|
117
152
|
if preview_lower.startswith((b"<?xml", b"<rss", b"<feed", b"<rdf")):
|
|
118
153
|
return stripped_content
|
|
119
154
|
|
|
120
|
-
if preview_lower.startswith(b"<!doctype html") or preview_lower.startswith(
|
|
155
|
+
if preview_lower.startswith(b"<!doctype html") or preview_lower.startswith(
|
|
156
|
+
b"<html"
|
|
157
|
+
):
|
|
121
158
|
raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
|
|
122
159
|
|
|
123
160
|
xml_start_patterns = (
|
|
@@ -148,15 +185,17 @@ def _fix_malformed_xml_bytes(content: bytes, actual_encoding: str = "utf-8") ->
|
|
|
148
185
|
content = _RE_DOUBLE_CLOSE_BYTES.sub(b"?>", content)
|
|
149
186
|
|
|
150
187
|
# Fix malformed attribute syntax like rss:version=2.0 (missing quotes)
|
|
151
|
-
content = _RE_UNQUOTED_ATTR_BYTES.sub(
|
|
188
|
+
content = _RE_UNQUOTED_ATTR_BYTES.sub(rb'\1="\2"', content)
|
|
152
189
|
|
|
153
190
|
# Update encoding in XML declaration to match actual encoding when a feed was transcoded.
|
|
154
191
|
if actual_encoding.lower() != "utf-16":
|
|
155
|
-
replacement =
|
|
192
|
+
replacement = (
|
|
193
|
+
rb"\1" + actual_encoding.encode("ascii", errors="replace") + rb"\3"
|
|
194
|
+
)
|
|
156
195
|
content = _RE_UTF16_ENCODING_BYTES.sub(replacement, content)
|
|
157
196
|
|
|
158
197
|
# Fix unclosed link tags - common in Atom feeds
|
|
159
|
-
content = _RE_UNCLOSED_LINK_BYTES.sub(
|
|
198
|
+
content = _RE_UNCLOSED_LINK_BYTES.sub(rb"<link\1/>", content)
|
|
160
199
|
|
|
161
200
|
return content
|
|
162
201
|
|
|
@@ -382,24 +421,26 @@ def _maybe_parse_json_feed(content: str | bytes) -> FastFeedParserDict | None:
|
|
|
382
421
|
return None
|
|
383
422
|
|
|
384
423
|
|
|
424
|
+
_STRICT_XML_PARSER = etree.XMLParser(
|
|
425
|
+
ns_clean=True,
|
|
426
|
+
recover=False,
|
|
427
|
+
collect_ids=False,
|
|
428
|
+
resolve_entities=False,
|
|
429
|
+
)
|
|
430
|
+
_RECOVER_XML_PARSER = etree.XMLParser(
|
|
431
|
+
ns_clean=True,
|
|
432
|
+
recover=True,
|
|
433
|
+
collect_ids=False,
|
|
434
|
+
resolve_entities=False,
|
|
435
|
+
)
|
|
436
|
+
|
|
437
|
+
|
|
385
438
|
def _parse_xml_root(xml_content: bytes) -> _Element:
|
|
386
439
|
try:
|
|
387
|
-
|
|
388
|
-
ns_clean=True,
|
|
389
|
-
recover=False,
|
|
390
|
-
collect_ids=False,
|
|
391
|
-
resolve_entities=False,
|
|
392
|
-
)
|
|
393
|
-
root = etree.fromstring(xml_content, parser=strict_parser)
|
|
440
|
+
root = etree.fromstring(xml_content, parser=_STRICT_XML_PARSER)
|
|
394
441
|
except etree.XMLSyntaxError:
|
|
395
|
-
recover_parser = etree.XMLParser(
|
|
396
|
-
ns_clean=True,
|
|
397
|
-
recover=True,
|
|
398
|
-
collect_ids=False,
|
|
399
|
-
resolve_entities=False,
|
|
400
|
-
)
|
|
401
442
|
try:
|
|
402
|
-
root = etree.fromstring(xml_content, parser=
|
|
443
|
+
root = etree.fromstring(xml_content, parser=_RECOVER_XML_PARSER)
|
|
403
444
|
except etree.XMLSyntaxError as e:
|
|
404
445
|
raise ValueError(f"Failed to parse XML content: {str(e)}")
|
|
405
446
|
|
|
@@ -486,16 +527,16 @@ def _raise_for_non_feed_root(
|
|
|
486
527
|
if base_msg is None:
|
|
487
528
|
return
|
|
488
529
|
|
|
489
|
-
error_msg =
|
|
530
|
+
error_msg = (
|
|
531
|
+
_extract_error_message(root, raw_bytes).strip()[:300] or "No error message"
|
|
532
|
+
)
|
|
490
533
|
|
|
491
534
|
if error_msg != "No error message" and len(error_msg) > 10:
|
|
492
535
|
raise ValueError(f"{base_msg}: {error_msg[:150]}")
|
|
493
536
|
raise ValueError(base_msg)
|
|
494
537
|
|
|
495
538
|
|
|
496
|
-
_RE_META_REFRESH_URL = re.compile(
|
|
497
|
-
r'url\s*=\s*["\']?\s*([^"\'>\s]+)', re.IGNORECASE
|
|
498
|
-
)
|
|
539
|
+
_RE_META_REFRESH_URL = re.compile(r'url\s*=\s*["\']?\s*([^"\'>\s]+)', re.IGNORECASE)
|
|
499
540
|
|
|
500
541
|
|
|
501
542
|
def _extract_meta_refresh_url(content: str | bytes, base_url: str) -> str | None:
|
|
@@ -544,7 +585,8 @@ def _detect_feed_structure(
|
|
|
544
585
|
if channel is None:
|
|
545
586
|
has_atom_elements = any(
|
|
546
587
|
isinstance(child.tag, str)
|
|
547
|
-
and child.tag
|
|
588
|
+
and child.tag
|
|
589
|
+
in {"entry", "title", "subtitle", "updated", "id", "author", "link"}
|
|
548
590
|
for child in root
|
|
549
591
|
)
|
|
550
592
|
if has_atom_elements:
|
|
@@ -572,7 +614,9 @@ def _detect_feed_structure(
|
|
|
572
614
|
items = []
|
|
573
615
|
items.append(child)
|
|
574
616
|
if not items:
|
|
575
|
-
items = channel.xpath(".//item") or channel.xpath(
|
|
617
|
+
items = channel.xpath(".//item") or channel.xpath(
|
|
618
|
+
".//*[local-name()='item']"
|
|
619
|
+
)
|
|
576
620
|
|
|
577
621
|
if not items:
|
|
578
622
|
items = channel.findall("entry")
|
|
@@ -619,7 +663,9 @@ def _detect_feed_structure(
|
|
|
619
663
|
if root.tag == "{http://www.w3.org/1999/02/22-rdf-syntax-ns#}RDF":
|
|
620
664
|
feed_type = "rdf"
|
|
621
665
|
channel = root
|
|
622
|
-
items = channel.findall(".//{http://purl.org/rss/1.0/}item") or channel.findall(
|
|
666
|
+
items = channel.findall(".//{http://purl.org/rss/1.0/}item") or channel.findall(
|
|
667
|
+
"item"
|
|
668
|
+
)
|
|
623
669
|
return feed_type, channel, items, atom_namespace
|
|
624
670
|
|
|
625
671
|
raise ValueError(f"Unknown feed type: {root.tag}")
|
|
@@ -642,6 +688,13 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
|
642
688
|
|
|
643
689
|
feed = _parse_feed_info(channel, feed_type, atom_namespace)
|
|
644
690
|
|
|
691
|
+
# Detect once whether media namespace is used anywhere in the document
|
|
692
|
+
has_media_ns = (
|
|
693
|
+
b"search.yahoo.com/mrss" in xml_content
|
|
694
|
+
if isinstance(xml_content, bytes)
|
|
695
|
+
else "search.yahoo.com/mrss" in xml_content
|
|
696
|
+
)
|
|
697
|
+
|
|
645
698
|
# Parse entries
|
|
646
699
|
entries: list[FastFeedParserDict] = []
|
|
647
700
|
feed["entries"] = entries
|
|
@@ -650,6 +703,7 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
|
650
703
|
item,
|
|
651
704
|
feed_type,
|
|
652
705
|
atom_namespace,
|
|
706
|
+
has_media_ns,
|
|
653
707
|
)
|
|
654
708
|
# Ensure that titles and descriptions are always present
|
|
655
709
|
entry["title"] = entry.get("title", "").strip()
|
|
@@ -674,6 +728,7 @@ def parse(source: str | bytes) -> FastFeedParserDict:
|
|
|
674
728
|
"""
|
|
675
729
|
is_url = isinstance(source, str) and source.startswith(("http://", "https://"))
|
|
676
730
|
if is_url:
|
|
731
|
+
assert isinstance(source, str)
|
|
677
732
|
content = _fetch_url_content(source)
|
|
678
733
|
else:
|
|
679
734
|
content = source
|
|
@@ -683,6 +738,7 @@ def parse(source: str | bytes) -> FastFeedParserDict:
|
|
|
683
738
|
except ValueError as e:
|
|
684
739
|
if not is_url:
|
|
685
740
|
raise
|
|
741
|
+
assert isinstance(source, str)
|
|
686
742
|
err_msg = str(e)
|
|
687
743
|
if "HTML" not in err_msg and "not a valid RSS/Atom feed" not in err_msg:
|
|
688
744
|
raise
|
|
@@ -912,7 +968,9 @@ def _coerce_int_fields(mapping: dict[str, Any], fields: tuple[str, ...]) -> None
|
|
|
912
968
|
mapping.pop(field, None)
|
|
913
969
|
|
|
914
970
|
|
|
915
|
-
def _populate_entry_links(
|
|
971
|
+
def _populate_entry_links(
|
|
972
|
+
entry: FastFeedParserDict, item: _Element, atom_ns: str
|
|
973
|
+
) -> None:
|
|
916
974
|
entry_links: list[dict[str, Optional[str]]] = []
|
|
917
975
|
alternate_link: Optional[dict[str, Optional[str]]] = None
|
|
918
976
|
for link in item.findall(f"{{{atom_ns}}}link"):
|
|
@@ -933,7 +991,9 @@ def _populate_entry_links(entry: FastFeedParserDict, item: _Element, atom_ns: st
|
|
|
933
991
|
|
|
934
992
|
guid = item.find("guid")
|
|
935
993
|
guid_text = guid.text.strip() if guid is not None and guid.text else None
|
|
936
|
-
is_guid_url = guid_text is not None and guid_text.startswith(
|
|
994
|
+
is_guid_url = guid_text is not None and guid_text.startswith(
|
|
995
|
+
("http://", "https://")
|
|
996
|
+
)
|
|
937
997
|
|
|
938
998
|
if is_guid_url and "link" not in entry:
|
|
939
999
|
entry["link"] = guid_text
|
|
@@ -974,7 +1034,9 @@ def _populate_entry_content(
|
|
|
974
1034
|
entry["content"] = [
|
|
975
1035
|
{
|
|
976
1036
|
"type": content_type,
|
|
977
|
-
"language": content_el.get(
|
|
1037
|
+
"language": content_el.get(
|
|
1038
|
+
"{http://www.w3.org/XML/1998/namespace}lang"
|
|
1039
|
+
),
|
|
978
1040
|
"base": content_el.get("{http://www.w3.org/XML/1998/namespace}base"),
|
|
979
1041
|
"value": content_value,
|
|
980
1042
|
}
|
|
@@ -998,7 +1060,15 @@ def _populate_entry_content(
|
|
|
998
1060
|
if "<" in content_value:
|
|
999
1061
|
content_value = _RE_HTML_TAGS.sub(" ", content_value[:2048])
|
|
1000
1062
|
content_value = _html_mod.unescape(content_value)
|
|
1001
|
-
|
|
1063
|
+
if (
|
|
1064
|
+
" " in content_value
|
|
1065
|
+
or "\n" in content_value
|
|
1066
|
+
or "\t" in content_value
|
|
1067
|
+
or "\r" in content_value
|
|
1068
|
+
):
|
|
1069
|
+
content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
|
|
1070
|
+
else:
|
|
1071
|
+
content_value = content_value.strip()
|
|
1002
1072
|
entry["description"] = content_value[:512]
|
|
1003
1073
|
|
|
1004
1074
|
|
|
@@ -1089,7 +1159,9 @@ def _parse_enclosures(item: _Element) -> list[dict[str, Any]] | None:
|
|
|
1089
1159
|
return enclosures or None
|
|
1090
1160
|
|
|
1091
1161
|
|
|
1092
|
-
def _build_rss_item_text_maps(
|
|
1162
|
+
def _build_rss_item_text_maps(
|
|
1163
|
+
item: _Element,
|
|
1164
|
+
) -> tuple[dict[str, Optional[str]], dict[str, Optional[str]]]:
|
|
1093
1165
|
by_local: dict[str, Optional[str]] = {}
|
|
1094
1166
|
by_full: dict[str, Optional[str]] = {}
|
|
1095
1167
|
for child in item:
|
|
@@ -1099,15 +1171,21 @@ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]],
|
|
|
1099
1171
|
text_value = child.text or None
|
|
1100
1172
|
if tag not in by_full:
|
|
1101
1173
|
by_full[tag] = text_value
|
|
1102
|
-
|
|
1103
|
-
if "
|
|
1104
|
-
local =
|
|
1174
|
+
# Fast path: ~80% of RSS tags have no namespace or colon prefix
|
|
1175
|
+
if "{" in tag:
|
|
1176
|
+
local = tag.rsplit("}", 1)[1].lower()
|
|
1177
|
+
elif ":" in tag:
|
|
1178
|
+
local = tag.split(":", 1)[1].lower()
|
|
1179
|
+
else:
|
|
1180
|
+
local = tag.lower()
|
|
1105
1181
|
if local not in by_local:
|
|
1106
1182
|
by_local[local] = text_value
|
|
1107
1183
|
return by_local, by_full
|
|
1108
1184
|
|
|
1109
1185
|
|
|
1110
|
-
def _first_non_empty(
|
|
1186
|
+
def _first_non_empty(
|
|
1187
|
+
mapping: dict[str, Optional[str]], keys: tuple[str, ...]
|
|
1188
|
+
) -> Optional[str]:
|
|
1111
1189
|
for key in keys:
|
|
1112
1190
|
value = mapping.get(key)
|
|
1113
1191
|
if value:
|
|
@@ -1118,6 +1196,7 @@ def _first_non_empty(mapping: dict[str, Optional[str]], keys: tuple[str, ...]) -
|
|
|
1118
1196
|
def _parse_rss_feed_entry_fast(
|
|
1119
1197
|
item: _Element,
|
|
1120
1198
|
atom_ns: str,
|
|
1199
|
+
has_media_ns: bool = True,
|
|
1121
1200
|
) -> FastFeedParserDict:
|
|
1122
1201
|
text_by_local, text_by_full = _build_rss_item_text_maps(item)
|
|
1123
1202
|
|
|
@@ -1141,19 +1220,27 @@ def _parse_rss_feed_entry_fast(
|
|
|
1141
1220
|
if link:
|
|
1142
1221
|
entry["link"] = link.strip()
|
|
1143
1222
|
|
|
1144
|
-
published_source = _first_non_empty(
|
|
1223
|
+
published_source = _first_non_empty(
|
|
1224
|
+
text_by_local, ("pubdate", "published", "issued", "date")
|
|
1225
|
+
)
|
|
1145
1226
|
if published_source:
|
|
1146
1227
|
published = _parse_date(published_source)
|
|
1147
1228
|
if published:
|
|
1148
1229
|
entry["published"] = published
|
|
1149
1230
|
|
|
1150
|
-
updated_source = _first_non_empty(
|
|
1231
|
+
updated_source = _first_non_empty(
|
|
1232
|
+
text_by_local, ("lastbuilddate", "updated", "modified")
|
|
1233
|
+
)
|
|
1151
1234
|
if updated_source:
|
|
1152
1235
|
updated = _parse_date(updated_source)
|
|
1153
1236
|
if updated:
|
|
1154
1237
|
entry["updated"] = updated
|
|
1155
1238
|
|
|
1156
|
-
if
|
|
1239
|
+
if (
|
|
1240
|
+
"published" not in entry
|
|
1241
|
+
and rss_guid
|
|
1242
|
+
and not rss_guid.startswith(("http://", "https://"))
|
|
1243
|
+
):
|
|
1157
1244
|
guid_date = _parse_date(rss_guid)
|
|
1158
1245
|
if guid_date:
|
|
1159
1246
|
entry["published"] = guid_date
|
|
@@ -1161,15 +1248,30 @@ def _parse_rss_feed_entry_fast(
|
|
|
1161
1248
|
if "updated" in entry and "published" not in entry:
|
|
1162
1249
|
entry["published"] = entry["updated"]
|
|
1163
1250
|
|
|
1164
|
-
|
|
1251
|
+
# Inline link population for RSS (avoids redundant findall/find for 98.8% of entries)
|
|
1252
|
+
atom_links = item.findall(f"{{{atom_ns}}}link")
|
|
1253
|
+
if atom_links:
|
|
1254
|
+
# Has atom:link elements - use full logic
|
|
1255
|
+
_populate_entry_links(entry, item, atom_ns)
|
|
1256
|
+
else:
|
|
1257
|
+
# Common RSS case: no atom:link elements
|
|
1258
|
+
entry["links"] = []
|
|
1259
|
+
if (
|
|
1260
|
+
"link" not in entry
|
|
1261
|
+
and rss_guid
|
|
1262
|
+
and rss_guid.startswith(("http://", "https://"))
|
|
1263
|
+
):
|
|
1264
|
+
entry["link"] = rss_guid
|
|
1265
|
+
|
|
1165
1266
|
if "id" not in entry and "link" in entry:
|
|
1166
1267
|
entry["id"] = entry["link"]
|
|
1167
1268
|
|
|
1168
1269
|
_populate_entry_content(entry, item, "rss", atom_ns)
|
|
1169
1270
|
|
|
1170
|
-
|
|
1171
|
-
|
|
1172
|
-
|
|
1271
|
+
if has_media_ns:
|
|
1272
|
+
media_contents = _parse_media_content(item)
|
|
1273
|
+
if media_contents:
|
|
1274
|
+
entry["media_content"] = media_contents
|
|
1173
1275
|
|
|
1174
1276
|
enclosures = _parse_enclosures(item)
|
|
1175
1277
|
if enclosures:
|
|
@@ -1178,7 +1280,11 @@ def _parse_rss_feed_entry_fast(
|
|
|
1178
1280
|
author = _first_non_empty(text_by_local, ("author", "creator"))
|
|
1179
1281
|
if not author:
|
|
1180
1282
|
atom_author = item.find(f"{{{atom_ns}}}author/{{{atom_ns}}}name")
|
|
1181
|
-
author =
|
|
1283
|
+
author = (
|
|
1284
|
+
atom_author.text.strip()
|
|
1285
|
+
if atom_author is not None and atom_author.text
|
|
1286
|
+
else None
|
|
1287
|
+
)
|
|
1182
1288
|
if author:
|
|
1183
1289
|
entry["author"] = author.strip()
|
|
1184
1290
|
|
|
@@ -1196,6 +1302,7 @@ def _parse_rss_feed_entry_fast(
|
|
|
1196
1302
|
def _parse_atom_feed_entry_fast(
|
|
1197
1303
|
item: _Element,
|
|
1198
1304
|
atom_ns: str,
|
|
1305
|
+
has_media_ns: bool = True,
|
|
1199
1306
|
) -> FastFeedParserDict:
|
|
1200
1307
|
ns = f"{{{atom_ns}}}"
|
|
1201
1308
|
entry = FastFeedParserDict()
|
|
@@ -1266,9 +1373,10 @@ def _parse_atom_feed_entry_fast(
|
|
|
1266
1373
|
|
|
1267
1374
|
_populate_entry_content(entry, item, "atom", atom_ns)
|
|
1268
1375
|
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
|
|
1376
|
+
if has_media_ns:
|
|
1377
|
+
media_contents = _parse_media_content(item)
|
|
1378
|
+
if media_contents:
|
|
1379
|
+
entry["media_content"] = media_contents
|
|
1272
1380
|
|
|
1273
1381
|
enclosures = _parse_enclosures(item)
|
|
1274
1382
|
if enclosures:
|
|
@@ -1290,15 +1398,16 @@ def _parse_feed_entry(
|
|
|
1290
1398
|
item: _Element,
|
|
1291
1399
|
feed_type: _FeedType,
|
|
1292
1400
|
atom_namespace: Optional[str] = None,
|
|
1401
|
+
has_media_ns: bool = True,
|
|
1293
1402
|
) -> FastFeedParserDict:
|
|
1294
1403
|
# Use dynamic atom namespace or fallback to default
|
|
1295
1404
|
atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
|
|
1296
1405
|
|
|
1297
1406
|
if feed_type == "rss":
|
|
1298
|
-
return _parse_rss_feed_entry_fast(item, atom_ns)
|
|
1407
|
+
return _parse_rss_feed_entry_fast(item, atom_ns, has_media_ns)
|
|
1299
1408
|
|
|
1300
1409
|
if feed_type == "atom":
|
|
1301
|
-
return _parse_atom_feed_entry_fast(item, atom_ns)
|
|
1410
|
+
return _parse_atom_feed_entry_fast(item, atom_ns, has_media_ns)
|
|
1302
1411
|
|
|
1303
1412
|
# RDF path uses the generic field machinery
|
|
1304
1413
|
# Check if this is Atom 0.3 to use different date field names
|
|
@@ -1392,7 +1501,11 @@ def _parse_feed_entry(
|
|
|
1392
1501
|
entry["updated"] = _parse_date(fallback_updated)
|
|
1393
1502
|
|
|
1394
1503
|
# Try to extract date from GUID as final fallback
|
|
1395
|
-
if
|
|
1504
|
+
if (
|
|
1505
|
+
"published" not in entry
|
|
1506
|
+
and rss_guid
|
|
1507
|
+
and not rss_guid.startswith(("http://", "https://"))
|
|
1508
|
+
):
|
|
1396
1509
|
guid_date = _parse_date(rss_guid)
|
|
1397
1510
|
if guid_date:
|
|
1398
1511
|
entry["published"] = guid_date
|
|
@@ -1412,9 +1525,10 @@ def _parse_feed_entry(
|
|
|
1412
1525
|
|
|
1413
1526
|
_populate_entry_content(entry, item, feed_type, atom_ns)
|
|
1414
1527
|
|
|
1415
|
-
|
|
1416
|
-
|
|
1417
|
-
|
|
1528
|
+
if has_media_ns:
|
|
1529
|
+
media_contents = _parse_media_content(item)
|
|
1530
|
+
if media_contents:
|
|
1531
|
+
entry["media_content"] = media_contents
|
|
1418
1532
|
|
|
1419
1533
|
enclosures = _parse_enclosures(item)
|
|
1420
1534
|
if enclosures:
|
|
@@ -1427,7 +1541,9 @@ def _parse_feed_entry(
|
|
|
1427
1541
|
False,
|
|
1428
1542
|
)
|
|
1429
1543
|
if not author:
|
|
1430
|
-
author = element_get(
|
|
1544
|
+
author = element_get(
|
|
1545
|
+
"{http://purl.org/dc/elements/1.1/}creator"
|
|
1546
|
+
) or element_get("author")
|
|
1431
1547
|
if author:
|
|
1432
1548
|
entry["author"] = author
|
|
1433
1549
|
|
|
@@ -1442,9 +1558,9 @@ def _parse_feed_entry(
|
|
|
1442
1558
|
def _field_value_getter(
|
|
1443
1559
|
root: _Element,
|
|
1444
1560
|
feed_type: _FeedType,
|
|
1445
|
-
cached_get: Optional[
|
|
1561
|
+
cached_get: Optional[_ElementValueGetter] = None,
|
|
1446
1562
|
) -> Callable[[str, str, str, bool], str | None]:
|
|
1447
|
-
get_value = cached_get or _cached_element_value_factory(root)
|
|
1563
|
+
get_value: _ElementValueGetter = cached_get or _cached_element_value_factory(root)
|
|
1448
1564
|
|
|
1449
1565
|
if feed_type == "rss":
|
|
1450
1566
|
|
|
@@ -1533,7 +1649,11 @@ def _get_element_value(
|
|
|
1533
1649
|
el = found
|
|
1534
1650
|
break
|
|
1535
1651
|
else:
|
|
1536
|
-
prefixed_paths = [
|
|
1652
|
+
prefixed_paths = [
|
|
1653
|
+
f"rss:{path_lower}",
|
|
1654
|
+
f"atom:{path_lower}",
|
|
1655
|
+
f"dc:{path_lower}",
|
|
1656
|
+
]
|
|
1537
1657
|
for child in root:
|
|
1538
1658
|
if not isinstance(child.tag, str):
|
|
1539
1659
|
continue
|
|
@@ -1553,7 +1673,7 @@ def _get_element_value(
|
|
|
1553
1673
|
|
|
1554
1674
|
def _cached_element_value_factory(
|
|
1555
1675
|
root: _Element,
|
|
1556
|
-
) ->
|
|
1676
|
+
) -> _ElementValueGetter:
|
|
1557
1677
|
"""Create a closure with a child tag index for fast namespace-prefix lookups."""
|
|
1558
1678
|
# Build child tag index once: O(children) instead of O(children × misses)
|
|
1559
1679
|
child_index: dict[str, _Element] = {}
|
|
@@ -1562,7 +1682,9 @@ def _cached_element_value_factory(
|
|
|
1562
1682
|
child_index[child.tag.lower()] = child
|
|
1563
1683
|
|
|
1564
1684
|
def getter(path: str, attribute: Optional[str] = None) -> Optional[str]:
|
|
1565
|
-
return _get_element_value(
|
|
1685
|
+
return _get_element_value(
|
|
1686
|
+
root, path, attribute=attribute, child_index=child_index
|
|
1687
|
+
)
|
|
1566
1688
|
|
|
1567
1689
|
return getter
|
|
1568
1690
|
|
|
@@ -1591,7 +1713,13 @@ def _normalize_iso_datetime_string(value: str) -> str:
|
|
|
1591
1713
|
if cleaned.endswith(("Z", "z")):
|
|
1592
1714
|
cleaned = cleaned[:-1] + "+00:00"
|
|
1593
1715
|
|
|
1594
|
-
if
|
|
1716
|
+
if (
|
|
1717
|
+
" " in cleaned
|
|
1718
|
+
and "T" not in cleaned[:11]
|
|
1719
|
+
and len(cleaned) >= 10
|
|
1720
|
+
and cleaned[4] == "-"
|
|
1721
|
+
and cleaned[0:4].isdigit()
|
|
1722
|
+
):
|
|
1595
1723
|
date_part, rest = cleaned.split(" ", 1)
|
|
1596
1724
|
if rest and rest[0].isdigit():
|
|
1597
1725
|
cleaned = f"{date_part}T{rest}"
|
|
@@ -1616,8 +1744,64 @@ def _ensure_utc(dt: datetime.datetime) -> Optional[datetime.datetime]:
|
|
|
1616
1744
|
return None
|
|
1617
1745
|
|
|
1618
1746
|
|
|
1747
|
+
def _fast_rfc822_to_iso(value: str) -> Optional[str]:
|
|
1748
|
+
"""Fast RFC-822 date to ISO string, bypassing datetime objects for UTC dates."""
|
|
1749
|
+
m = _RE_RFC822.match(value)
|
|
1750
|
+
if not m:
|
|
1751
|
+
return None
|
|
1752
|
+
day, mon_str, year, hour, minute, second, tz = m.groups()
|
|
1753
|
+
month = _MONTHS_RFC822.get(mon_str.lower())
|
|
1754
|
+
if month is None:
|
|
1755
|
+
return None
|
|
1756
|
+
if tz[0] in "+-":
|
|
1757
|
+
tz_offset_seconds = (int(tz[1:3]) * 3600 + int(tz[3:5]) * 60) * (
|
|
1758
|
+
1 if tz[0] == "+" else -1
|
|
1759
|
+
)
|
|
1760
|
+
else:
|
|
1761
|
+
tz_offset_seconds = _TZ_OFFSETS_RFC822.get(tz)
|
|
1762
|
+
if tz_offset_seconds is None:
|
|
1763
|
+
return None # Unknown tz name, fall through to full parser
|
|
1764
|
+
# Python requires offset strictly between -24h and +24h
|
|
1765
|
+
if not (-86400 < tz_offset_seconds < 86400):
|
|
1766
|
+
return None
|
|
1767
|
+
d = int(day)
|
|
1768
|
+
h = int(hour)
|
|
1769
|
+
mi = int(minute)
|
|
1770
|
+
s = int(second)
|
|
1771
|
+
# Hour 24 is invalid (even ISO only allows 24:00:00); roll to next day at 00:mm:ss
|
|
1772
|
+
if h == 24:
|
|
1773
|
+
base = datetime.date(int(year), month, d) + datetime.timedelta(days=1)
|
|
1774
|
+
h = 0
|
|
1775
|
+
if tz_offset_seconds == 0:
|
|
1776
|
+
return f"{base.year:04d}-{base.month:02d}-{base.day:02d}T{h:02d}:{mi:02d}:{s:02d}+00:00"
|
|
1777
|
+
dt = datetime.datetime(
|
|
1778
|
+
base.year,
|
|
1779
|
+
base.month,
|
|
1780
|
+
base.day,
|
|
1781
|
+
h,
|
|
1782
|
+
mi,
|
|
1783
|
+
s,
|
|
1784
|
+
tzinfo=datetime.timezone(datetime.timedelta(seconds=tz_offset_seconds)),
|
|
1785
|
+
)
|
|
1786
|
+
utc = dt.astimezone(_UTC)
|
|
1787
|
+
return f"{utc.year:04d}-{utc.month:02d}-{utc.day:02d}T{utc.hour:02d}:{utc.minute:02d}:{utc.second:02d}+00:00"
|
|
1788
|
+
if tz_offset_seconds == 0:
|
|
1789
|
+
return f"{year}-{month:02d}-{d:02d}T{hour}:{minute}:{second}+00:00"
|
|
1790
|
+
dt = datetime.datetime(
|
|
1791
|
+
int(year),
|
|
1792
|
+
month,
|
|
1793
|
+
d,
|
|
1794
|
+
h,
|
|
1795
|
+
mi,
|
|
1796
|
+
s,
|
|
1797
|
+
tzinfo=datetime.timezone(datetime.timedelta(seconds=tz_offset_seconds)),
|
|
1798
|
+
)
|
|
1799
|
+
utc = dt.astimezone(_UTC)
|
|
1800
|
+
return f"{utc.year:04d}-{utc.month:02d}-{utc.day:02d}T{utc.hour:02d}:{utc.minute:02d}:{utc.second:02d}+00:00"
|
|
1801
|
+
|
|
1802
|
+
|
|
1619
1803
|
def _parsedate_to_utc(value: str) -> Optional[datetime.datetime]:
|
|
1620
|
-
"""
|
|
1804
|
+
"""RFC-822 / RFC-2822 parsing via email.utils (fallback)."""
|
|
1621
1805
|
try:
|
|
1622
1806
|
parsed = parsedate_to_datetime(value)
|
|
1623
1807
|
except (TypeError, ValueError, IndexError):
|
|
@@ -1690,7 +1874,9 @@ def _slow_dateparser(value: str) -> Optional[datetime.datetime]:
|
|
|
1690
1874
|
except ImportError:
|
|
1691
1875
|
return None
|
|
1692
1876
|
try:
|
|
1693
|
-
return _dateparser.parse(
|
|
1877
|
+
return _dateparser.parse(
|
|
1878
|
+
value, languages=["en"], settings={**_DATEPARSER_SETTINGS}
|
|
1879
|
+
)
|
|
1694
1880
|
except (ValueError, TypeError):
|
|
1695
1881
|
return None
|
|
1696
1882
|
|
|
@@ -1717,8 +1903,9 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1717
1903
|
last = candidate[-1]
|
|
1718
1904
|
# Most common: ends with 'Z' (e.g., 2024-01-15T10:30:00Z)
|
|
1719
1905
|
if last in ("Z", "z"):
|
|
1906
|
+
iso = candidate[:-1] + "+00:00"
|
|
1720
1907
|
try:
|
|
1721
|
-
dt = datetime.datetime.fromisoformat(
|
|
1908
|
+
dt = datetime.datetime.fromisoformat(iso)
|
|
1722
1909
|
return dt.isoformat()
|
|
1723
1910
|
except ValueError:
|
|
1724
1911
|
pass # Fall through to full parsing
|
|
@@ -1726,7 +1913,9 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1726
1913
|
elif clen > 6 and candidate[-6] in ("+", "-") and candidate[-3] == ":":
|
|
1727
1914
|
try:
|
|
1728
1915
|
dt = datetime.datetime.fromisoformat(candidate)
|
|
1729
|
-
|
|
1916
|
+
if dt.tzinfo is _UTC:
|
|
1917
|
+
return dt.isoformat()
|
|
1918
|
+
utc_dt = dt.astimezone(_UTC)
|
|
1730
1919
|
return utc_dt.isoformat()
|
|
1731
1920
|
except (ValueError, OverflowError):
|
|
1732
1921
|
pass # Fall through to full parsing
|
|
@@ -1743,14 +1932,23 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1743
1932
|
if not ((year % 4 == 0 and year % 100 != 0) or (year % 400 == 0)):
|
|
1744
1933
|
candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
|
|
1745
1934
|
|
|
1746
|
-
if "24:
|
|
1747
|
-
|
|
1748
|
-
|
|
1749
|
-
|
|
1935
|
+
if "T24:" in candidate or " 24:" in candidate:
|
|
1936
|
+
m24 = re.search(r"(\d{4}-\d{2}-\d{2})[T ]24:(\d{2}):(\d{2})", candidate)
|
|
1937
|
+
if m24:
|
|
1938
|
+
base = datetime.date.fromisoformat(m24.group(1))
|
|
1939
|
+
mins, secs = int(m24.group(2)), int(m24.group(3))
|
|
1940
|
+
next_day = base + datetime.timedelta(days=1)
|
|
1941
|
+
candidate = (
|
|
1942
|
+
candidate[: m24.start()]
|
|
1943
|
+
+ f"{next_day}T00:{mins:02d}:{secs:02d}"
|
|
1944
|
+
+ candidate[m24.end() :]
|
|
1945
|
+
)
|
|
1750
1946
|
|
|
1751
1947
|
dt: Optional[datetime.datetime] = None
|
|
1752
1948
|
|
|
1753
|
-
is_iso_like =
|
|
1949
|
+
is_iso_like = (
|
|
1950
|
+
len(candidate) >= 10 and candidate[4] == "-" and candidate[0:4].isdigit()
|
|
1951
|
+
)
|
|
1754
1952
|
if is_iso_like:
|
|
1755
1953
|
iso_candidate = _normalize_iso_datetime_string(candidate)
|
|
1756
1954
|
try:
|
|
@@ -1762,6 +1960,10 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1762
1960
|
if utc_dt is not None:
|
|
1763
1961
|
return utc_dt.isoformat()
|
|
1764
1962
|
|
|
1963
|
+
rfc822_result = _fast_rfc822_to_iso(candidate)
|
|
1964
|
+
if rfc822_result is not None:
|
|
1965
|
+
return rfc822_result
|
|
1966
|
+
|
|
1765
1967
|
dt = _parsedate_to_utc(candidate)
|
|
1766
1968
|
if dt is not None:
|
|
1767
1969
|
return dt.isoformat()
|
|
@@ -34,12 +34,18 @@ def test_parse_bytes_with_non_utf8_encoding():
|
|
|
34
34
|
|
|
35
35
|
def test_meta_refresh_extraction():
|
|
36
36
|
html = '<!doctype html><html><head><meta http-equiv=refresh content="0; url=https://example.com/feed.xml"></head></html>'
|
|
37
|
-
assert
|
|
37
|
+
assert (
|
|
38
|
+
_extract_meta_refresh_url(html, "https://example.com/feed/")
|
|
39
|
+
== "https://example.com/feed.xml"
|
|
40
|
+
)
|
|
38
41
|
|
|
39
42
|
|
|
40
43
|
def test_meta_refresh_relative_url():
|
|
41
44
|
html = b'<html><head><meta http-equiv="refresh" content="0;url=/index.xml"></head></html>'
|
|
42
|
-
assert
|
|
45
|
+
assert (
|
|
46
|
+
_extract_meta_refresh_url(html, "https://example.com/feed/")
|
|
47
|
+
== "https://example.com/index.xml"
|
|
48
|
+
)
|
|
43
49
|
|
|
44
50
|
|
|
45
51
|
def test_meta_refresh_none_when_missing():
|
|
@@ -50,4 +56,3 @@ def test_meta_refresh_none_when_missing():
|
|
|
50
56
|
def test_meta_refresh_none_when_same_url():
|
|
51
57
|
html = '<html><head><meta http-equiv="refresh" content="0; url=https://example.com/"></head></html>'
|
|
52
58
|
assert _extract_meta_refresh_url(html, "https://example.com/") is None
|
|
53
|
-
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{fastfeedparser-0.5.2 → fastfeedparser-0.5.4}/src/fastfeedparser.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|