fastfeedparser 0.5.3__tar.gz → 0.5.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fastfeedparser-0.5.3/src/fastfeedparser.egg-info → fastfeedparser-0.5.4}/PKG-INFO +1 -1
- fastfeedparser-0.5.4/pyproject.toml +23 -0
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.4}/setup.cfg +1 -1
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.4}/src/fastfeedparser/main.py +153 -48
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.4/src/fastfeedparser.egg-info}/PKG-INFO +1 -1
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.4}/tests/test_encoding.py +8 -3
- fastfeedparser-0.5.3/pyproject.toml +0 -7
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.4}/LICENSE +0 -0
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.4}/README.md +0 -0
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.4}/src/fastfeedparser/__init__.py +0 -0
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.4}/src/fastfeedparser.egg-info/SOURCES.txt +0 -0
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.4}/src/fastfeedparser.egg-info/dependency_links.txt +0 -0
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.4}/src/fastfeedparser.egg-info/requires.txt +0 -0
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.4}/src/fastfeedparser.egg-info/top_level.txt +0 -0
- {fastfeedparser-0.5.3 → fastfeedparser-0.5.4}/tests/test_integration.py +0 -0
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools~=67.0", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[tool.ruff]
|
|
6
|
+
extend-exclude = [
|
|
7
|
+
"1.py",
|
|
8
|
+
"debug_*.py",
|
|
9
|
+
"test_*.py",
|
|
10
|
+
"benchmark.py",
|
|
11
|
+
"investigate_failures.py",
|
|
12
|
+
"show_error_messages.py",
|
|
13
|
+
"check_oh4_dates.py",
|
|
14
|
+
"comprehensive_debug.py",
|
|
15
|
+
"profile_dylanharris.py",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
[tool.ty.rules]
|
|
19
|
+
unresolved-import = "ignore"
|
|
20
|
+
|
|
21
|
+
[tool.pytest.ini_options]
|
|
22
|
+
testpaths = ["tests"]
|
|
23
|
+
|
|
@@ -15,7 +15,7 @@ try:
|
|
|
15
15
|
HAS_BROTLI = True
|
|
16
16
|
except ImportError:
|
|
17
17
|
HAS_BROTLI = False
|
|
18
|
-
from typing import Any, Callable, Optional, TYPE_CHECKING, Literal
|
|
18
|
+
from typing import Any, Callable, Optional, Protocol, TYPE_CHECKING, Literal
|
|
19
19
|
from urllib.parse import urljoin
|
|
20
20
|
from urllib.request import (
|
|
21
21
|
HTTPErrorProcessor,
|
|
@@ -32,6 +32,11 @@ if TYPE_CHECKING:
|
|
|
32
32
|
|
|
33
33
|
_FeedType = Literal["rss", "atom", "rdf"]
|
|
34
34
|
|
|
35
|
+
|
|
36
|
+
class _ElementValueGetter(Protocol):
|
|
37
|
+
def __call__(self, path: str, attribute: Optional[str] = None) -> Optional[str]: ...
|
|
38
|
+
|
|
39
|
+
|
|
35
40
|
_UTC = datetime.timezone.utc
|
|
36
41
|
|
|
37
42
|
# Pre-compiled regex patterns for performance
|
|
@@ -39,16 +44,16 @@ _RE_XML_DECL_ENCODING = re.compile(
|
|
|
39
44
|
r'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
|
|
40
45
|
)
|
|
41
46
|
_RE_XML_DECL_ENCODING_BYTES = re.compile(
|
|
42
|
-
|
|
47
|
+
rb'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
|
|
43
48
|
)
|
|
44
|
-
_RE_DOUBLE_XML_DECL_BYTES = re.compile(
|
|
45
|
-
_RE_DOUBLE_CLOSE_BYTES = re.compile(
|
|
46
|
-
_RE_UNQUOTED_ATTR_BYTES = re.compile(
|
|
49
|
+
_RE_DOUBLE_XML_DECL_BYTES = re.compile(rb"<\?xml\?xml\s+", re.IGNORECASE)
|
|
50
|
+
_RE_DOUBLE_CLOSE_BYTES = re.compile(rb"\?\?>\s*")
|
|
51
|
+
_RE_UNQUOTED_ATTR_BYTES = re.compile(rb'(\s+[\w:]+)=([^\s>"\']+)')
|
|
47
52
|
_RE_UTF16_ENCODING_BYTES = re.compile(
|
|
48
|
-
|
|
53
|
+
rb'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
|
|
49
54
|
)
|
|
50
55
|
_RE_UNCLOSED_LINK_BYTES = re.compile(
|
|
51
|
-
|
|
56
|
+
rb"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
|
|
52
57
|
)
|
|
53
58
|
_RE_FEB29 = re.compile(r"(\d{4})-02-29")
|
|
54
59
|
_RE_HTML_TAGS = re.compile(r"<[^>]+>")
|
|
@@ -60,13 +65,31 @@ _RE_RFC822 = re.compile(
|
|
|
60
65
|
r"(?:\w{3},\s+)?(\d{1,2})\s+(\w{3})\s+(\d{4})\s+(\d{2}):(\d{2}):(\d{2})\s+([+-]\d{4}|[A-Z]{2,5})"
|
|
61
66
|
)
|
|
62
67
|
_MONTHS_RFC822: dict[str, int] = {
|
|
63
|
-
"jan": 1,
|
|
64
|
-
"
|
|
68
|
+
"jan": 1,
|
|
69
|
+
"feb": 2,
|
|
70
|
+
"mar": 3,
|
|
71
|
+
"apr": 4,
|
|
72
|
+
"may": 5,
|
|
73
|
+
"jun": 6,
|
|
74
|
+
"jul": 7,
|
|
75
|
+
"aug": 8,
|
|
76
|
+
"sep": 9,
|
|
77
|
+
"oct": 10,
|
|
78
|
+
"nov": 11,
|
|
79
|
+
"dec": 12,
|
|
65
80
|
}
|
|
66
81
|
_TZ_OFFSETS_RFC822: dict[str, int] = {
|
|
67
|
-
"GMT": 0,
|
|
68
|
-
"
|
|
69
|
-
"
|
|
82
|
+
"GMT": 0,
|
|
83
|
+
"UTC": 0,
|
|
84
|
+
"UT": 0,
|
|
85
|
+
"EST": -18000,
|
|
86
|
+
"EDT": -14400,
|
|
87
|
+
"CST": -21600,
|
|
88
|
+
"CDT": -18000,
|
|
89
|
+
"MST": -25200,
|
|
90
|
+
"MDT": -21600,
|
|
91
|
+
"PST": -28800,
|
|
92
|
+
"PDT": -25200,
|
|
70
93
|
}
|
|
71
94
|
|
|
72
95
|
|
|
@@ -129,7 +152,9 @@ def _clean_feed_bytes(content: bytes) -> bytes:
|
|
|
129
152
|
if preview_lower.startswith((b"<?xml", b"<rss", b"<feed", b"<rdf")):
|
|
130
153
|
return stripped_content
|
|
131
154
|
|
|
132
|
-
if preview_lower.startswith(b"<!doctype html") or preview_lower.startswith(
|
|
155
|
+
if preview_lower.startswith(b"<!doctype html") or preview_lower.startswith(
|
|
156
|
+
b"<html"
|
|
157
|
+
):
|
|
133
158
|
raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
|
|
134
159
|
|
|
135
160
|
xml_start_patterns = (
|
|
@@ -160,15 +185,17 @@ def _fix_malformed_xml_bytes(content: bytes, actual_encoding: str = "utf-8") ->
|
|
|
160
185
|
content = _RE_DOUBLE_CLOSE_BYTES.sub(b"?>", content)
|
|
161
186
|
|
|
162
187
|
# Fix malformed attribute syntax like rss:version=2.0 (missing quotes)
|
|
163
|
-
content = _RE_UNQUOTED_ATTR_BYTES.sub(
|
|
188
|
+
content = _RE_UNQUOTED_ATTR_BYTES.sub(rb'\1="\2"', content)
|
|
164
189
|
|
|
165
190
|
# Update encoding in XML declaration to match actual encoding when a feed was transcoded.
|
|
166
191
|
if actual_encoding.lower() != "utf-16":
|
|
167
|
-
replacement =
|
|
192
|
+
replacement = (
|
|
193
|
+
rb"\1" + actual_encoding.encode("ascii", errors="replace") + rb"\3"
|
|
194
|
+
)
|
|
168
195
|
content = _RE_UTF16_ENCODING_BYTES.sub(replacement, content)
|
|
169
196
|
|
|
170
197
|
# Fix unclosed link tags - common in Atom feeds
|
|
171
|
-
content = _RE_UNCLOSED_LINK_BYTES.sub(
|
|
198
|
+
content = _RE_UNCLOSED_LINK_BYTES.sub(rb"<link\1/>", content)
|
|
172
199
|
|
|
173
200
|
return content
|
|
174
201
|
|
|
@@ -500,16 +527,16 @@ def _raise_for_non_feed_root(
|
|
|
500
527
|
if base_msg is None:
|
|
501
528
|
return
|
|
502
529
|
|
|
503
|
-
error_msg =
|
|
530
|
+
error_msg = (
|
|
531
|
+
_extract_error_message(root, raw_bytes).strip()[:300] or "No error message"
|
|
532
|
+
)
|
|
504
533
|
|
|
505
534
|
if error_msg != "No error message" and len(error_msg) > 10:
|
|
506
535
|
raise ValueError(f"{base_msg}: {error_msg[:150]}")
|
|
507
536
|
raise ValueError(base_msg)
|
|
508
537
|
|
|
509
538
|
|
|
510
|
-
_RE_META_REFRESH_URL = re.compile(
|
|
511
|
-
r'url\s*=\s*["\']?\s*([^"\'>\s]+)', re.IGNORECASE
|
|
512
|
-
)
|
|
539
|
+
_RE_META_REFRESH_URL = re.compile(r'url\s*=\s*["\']?\s*([^"\'>\s]+)', re.IGNORECASE)
|
|
513
540
|
|
|
514
541
|
|
|
515
542
|
def _extract_meta_refresh_url(content: str | bytes, base_url: str) -> str | None:
|
|
@@ -558,7 +585,8 @@ def _detect_feed_structure(
|
|
|
558
585
|
if channel is None:
|
|
559
586
|
has_atom_elements = any(
|
|
560
587
|
isinstance(child.tag, str)
|
|
561
|
-
and child.tag
|
|
588
|
+
and child.tag
|
|
589
|
+
in {"entry", "title", "subtitle", "updated", "id", "author", "link"}
|
|
562
590
|
for child in root
|
|
563
591
|
)
|
|
564
592
|
if has_atom_elements:
|
|
@@ -586,7 +614,9 @@ def _detect_feed_structure(
|
|
|
586
614
|
items = []
|
|
587
615
|
items.append(child)
|
|
588
616
|
if not items:
|
|
589
|
-
items = channel.xpath(".//item") or channel.xpath(
|
|
617
|
+
items = channel.xpath(".//item") or channel.xpath(
|
|
618
|
+
".//*[local-name()='item']"
|
|
619
|
+
)
|
|
590
620
|
|
|
591
621
|
if not items:
|
|
592
622
|
items = channel.findall("entry")
|
|
@@ -633,7 +663,9 @@ def _detect_feed_structure(
|
|
|
633
663
|
if root.tag == "{http://www.w3.org/1999/02/22-rdf-syntax-ns#}RDF":
|
|
634
664
|
feed_type = "rdf"
|
|
635
665
|
channel = root
|
|
636
|
-
items = channel.findall(".//{http://purl.org/rss/1.0/}item") or channel.findall(
|
|
666
|
+
items = channel.findall(".//{http://purl.org/rss/1.0/}item") or channel.findall(
|
|
667
|
+
"item"
|
|
668
|
+
)
|
|
637
669
|
return feed_type, channel, items, atom_namespace
|
|
638
670
|
|
|
639
671
|
raise ValueError(f"Unknown feed type: {root.tag}")
|
|
@@ -657,7 +689,11 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
|
657
689
|
feed = _parse_feed_info(channel, feed_type, atom_namespace)
|
|
658
690
|
|
|
659
691
|
# Detect once whether media namespace is used anywhere in the document
|
|
660
|
-
has_media_ns =
|
|
692
|
+
has_media_ns = (
|
|
693
|
+
b"search.yahoo.com/mrss" in xml_content
|
|
694
|
+
if isinstance(xml_content, bytes)
|
|
695
|
+
else "search.yahoo.com/mrss" in xml_content
|
|
696
|
+
)
|
|
661
697
|
|
|
662
698
|
# Parse entries
|
|
663
699
|
entries: list[FastFeedParserDict] = []
|
|
@@ -692,6 +728,7 @@ def parse(source: str | bytes) -> FastFeedParserDict:
|
|
|
692
728
|
"""
|
|
693
729
|
is_url = isinstance(source, str) and source.startswith(("http://", "https://"))
|
|
694
730
|
if is_url:
|
|
731
|
+
assert isinstance(source, str)
|
|
695
732
|
content = _fetch_url_content(source)
|
|
696
733
|
else:
|
|
697
734
|
content = source
|
|
@@ -701,6 +738,7 @@ def parse(source: str | bytes) -> FastFeedParserDict:
|
|
|
701
738
|
except ValueError as e:
|
|
702
739
|
if not is_url:
|
|
703
740
|
raise
|
|
741
|
+
assert isinstance(source, str)
|
|
704
742
|
err_msg = str(e)
|
|
705
743
|
if "HTML" not in err_msg and "not a valid RSS/Atom feed" not in err_msg:
|
|
706
744
|
raise
|
|
@@ -930,7 +968,9 @@ def _coerce_int_fields(mapping: dict[str, Any], fields: tuple[str, ...]) -> None
|
|
|
930
968
|
mapping.pop(field, None)
|
|
931
969
|
|
|
932
970
|
|
|
933
|
-
def _populate_entry_links(
|
|
971
|
+
def _populate_entry_links(
|
|
972
|
+
entry: FastFeedParserDict, item: _Element, atom_ns: str
|
|
973
|
+
) -> None:
|
|
934
974
|
entry_links: list[dict[str, Optional[str]]] = []
|
|
935
975
|
alternate_link: Optional[dict[str, Optional[str]]] = None
|
|
936
976
|
for link in item.findall(f"{{{atom_ns}}}link"):
|
|
@@ -951,7 +991,9 @@ def _populate_entry_links(entry: FastFeedParserDict, item: _Element, atom_ns: st
|
|
|
951
991
|
|
|
952
992
|
guid = item.find("guid")
|
|
953
993
|
guid_text = guid.text.strip() if guid is not None and guid.text else None
|
|
954
|
-
is_guid_url = guid_text is not None and guid_text.startswith(
|
|
994
|
+
is_guid_url = guid_text is not None and guid_text.startswith(
|
|
995
|
+
("http://", "https://")
|
|
996
|
+
)
|
|
955
997
|
|
|
956
998
|
if is_guid_url and "link" not in entry:
|
|
957
999
|
entry["link"] = guid_text
|
|
@@ -992,7 +1034,9 @@ def _populate_entry_content(
|
|
|
992
1034
|
entry["content"] = [
|
|
993
1035
|
{
|
|
994
1036
|
"type": content_type,
|
|
995
|
-
"language": content_el.get(
|
|
1037
|
+
"language": content_el.get(
|
|
1038
|
+
"{http://www.w3.org/XML/1998/namespace}lang"
|
|
1039
|
+
),
|
|
996
1040
|
"base": content_el.get("{http://www.w3.org/XML/1998/namespace}base"),
|
|
997
1041
|
"value": content_value,
|
|
998
1042
|
}
|
|
@@ -1016,7 +1060,12 @@ def _populate_entry_content(
|
|
|
1016
1060
|
if "<" in content_value:
|
|
1017
1061
|
content_value = _RE_HTML_TAGS.sub(" ", content_value[:2048])
|
|
1018
1062
|
content_value = _html_mod.unescape(content_value)
|
|
1019
|
-
if
|
|
1063
|
+
if (
|
|
1064
|
+
" " in content_value
|
|
1065
|
+
or "\n" in content_value
|
|
1066
|
+
or "\t" in content_value
|
|
1067
|
+
or "\r" in content_value
|
|
1068
|
+
):
|
|
1020
1069
|
content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
|
|
1021
1070
|
else:
|
|
1022
1071
|
content_value = content_value.strip()
|
|
@@ -1110,7 +1159,9 @@ def _parse_enclosures(item: _Element) -> list[dict[str, Any]] | None:
|
|
|
1110
1159
|
return enclosures or None
|
|
1111
1160
|
|
|
1112
1161
|
|
|
1113
|
-
def _build_rss_item_text_maps(
|
|
1162
|
+
def _build_rss_item_text_maps(
|
|
1163
|
+
item: _Element,
|
|
1164
|
+
) -> tuple[dict[str, Optional[str]], dict[str, Optional[str]]]:
|
|
1114
1165
|
by_local: dict[str, Optional[str]] = {}
|
|
1115
1166
|
by_full: dict[str, Optional[str]] = {}
|
|
1116
1167
|
for child in item:
|
|
@@ -1132,7 +1183,9 @@ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]],
|
|
|
1132
1183
|
return by_local, by_full
|
|
1133
1184
|
|
|
1134
1185
|
|
|
1135
|
-
def _first_non_empty(
|
|
1186
|
+
def _first_non_empty(
|
|
1187
|
+
mapping: dict[str, Optional[str]], keys: tuple[str, ...]
|
|
1188
|
+
) -> Optional[str]:
|
|
1136
1189
|
for key in keys:
|
|
1137
1190
|
value = mapping.get(key)
|
|
1138
1191
|
if value:
|
|
@@ -1167,19 +1220,27 @@ def _parse_rss_feed_entry_fast(
|
|
|
1167
1220
|
if link:
|
|
1168
1221
|
entry["link"] = link.strip()
|
|
1169
1222
|
|
|
1170
|
-
published_source = _first_non_empty(
|
|
1223
|
+
published_source = _first_non_empty(
|
|
1224
|
+
text_by_local, ("pubdate", "published", "issued", "date")
|
|
1225
|
+
)
|
|
1171
1226
|
if published_source:
|
|
1172
1227
|
published = _parse_date(published_source)
|
|
1173
1228
|
if published:
|
|
1174
1229
|
entry["published"] = published
|
|
1175
1230
|
|
|
1176
|
-
updated_source = _first_non_empty(
|
|
1231
|
+
updated_source = _first_non_empty(
|
|
1232
|
+
text_by_local, ("lastbuilddate", "updated", "modified")
|
|
1233
|
+
)
|
|
1177
1234
|
if updated_source:
|
|
1178
1235
|
updated = _parse_date(updated_source)
|
|
1179
1236
|
if updated:
|
|
1180
1237
|
entry["updated"] = updated
|
|
1181
1238
|
|
|
1182
|
-
if
|
|
1239
|
+
if (
|
|
1240
|
+
"published" not in entry
|
|
1241
|
+
and rss_guid
|
|
1242
|
+
and not rss_guid.startswith(("http://", "https://"))
|
|
1243
|
+
):
|
|
1183
1244
|
guid_date = _parse_date(rss_guid)
|
|
1184
1245
|
if guid_date:
|
|
1185
1246
|
entry["published"] = guid_date
|
|
@@ -1195,7 +1256,11 @@ def _parse_rss_feed_entry_fast(
|
|
|
1195
1256
|
else:
|
|
1196
1257
|
# Common RSS case: no atom:link elements
|
|
1197
1258
|
entry["links"] = []
|
|
1198
|
-
if
|
|
1259
|
+
if (
|
|
1260
|
+
"link" not in entry
|
|
1261
|
+
and rss_guid
|
|
1262
|
+
and rss_guid.startswith(("http://", "https://"))
|
|
1263
|
+
):
|
|
1199
1264
|
entry["link"] = rss_guid
|
|
1200
1265
|
|
|
1201
1266
|
if "id" not in entry and "link" in entry:
|
|
@@ -1215,7 +1280,11 @@ def _parse_rss_feed_entry_fast(
|
|
|
1215
1280
|
author = _first_non_empty(text_by_local, ("author", "creator"))
|
|
1216
1281
|
if not author:
|
|
1217
1282
|
atom_author = item.find(f"{{{atom_ns}}}author/{{{atom_ns}}}name")
|
|
1218
|
-
author =
|
|
1283
|
+
author = (
|
|
1284
|
+
atom_author.text.strip()
|
|
1285
|
+
if atom_author is not None and atom_author.text
|
|
1286
|
+
else None
|
|
1287
|
+
)
|
|
1219
1288
|
if author:
|
|
1220
1289
|
entry["author"] = author.strip()
|
|
1221
1290
|
|
|
@@ -1432,7 +1501,11 @@ def _parse_feed_entry(
|
|
|
1432
1501
|
entry["updated"] = _parse_date(fallback_updated)
|
|
1433
1502
|
|
|
1434
1503
|
# Try to extract date from GUID as final fallback
|
|
1435
|
-
if
|
|
1504
|
+
if (
|
|
1505
|
+
"published" not in entry
|
|
1506
|
+
and rss_guid
|
|
1507
|
+
and not rss_guid.startswith(("http://", "https://"))
|
|
1508
|
+
):
|
|
1436
1509
|
guid_date = _parse_date(rss_guid)
|
|
1437
1510
|
if guid_date:
|
|
1438
1511
|
entry["published"] = guid_date
|
|
@@ -1468,7 +1541,9 @@ def _parse_feed_entry(
|
|
|
1468
1541
|
False,
|
|
1469
1542
|
)
|
|
1470
1543
|
if not author:
|
|
1471
|
-
author = element_get(
|
|
1544
|
+
author = element_get(
|
|
1545
|
+
"{http://purl.org/dc/elements/1.1/}creator"
|
|
1546
|
+
) or element_get("author")
|
|
1472
1547
|
if author:
|
|
1473
1548
|
entry["author"] = author
|
|
1474
1549
|
|
|
@@ -1483,9 +1558,9 @@ def _parse_feed_entry(
|
|
|
1483
1558
|
def _field_value_getter(
|
|
1484
1559
|
root: _Element,
|
|
1485
1560
|
feed_type: _FeedType,
|
|
1486
|
-
cached_get: Optional[
|
|
1561
|
+
cached_get: Optional[_ElementValueGetter] = None,
|
|
1487
1562
|
) -> Callable[[str, str, str, bool], str | None]:
|
|
1488
|
-
get_value = cached_get or _cached_element_value_factory(root)
|
|
1563
|
+
get_value: _ElementValueGetter = cached_get or _cached_element_value_factory(root)
|
|
1489
1564
|
|
|
1490
1565
|
if feed_type == "rss":
|
|
1491
1566
|
|
|
@@ -1574,7 +1649,11 @@ def _get_element_value(
|
|
|
1574
1649
|
el = found
|
|
1575
1650
|
break
|
|
1576
1651
|
else:
|
|
1577
|
-
prefixed_paths = [
|
|
1652
|
+
prefixed_paths = [
|
|
1653
|
+
f"rss:{path_lower}",
|
|
1654
|
+
f"atom:{path_lower}",
|
|
1655
|
+
f"dc:{path_lower}",
|
|
1656
|
+
]
|
|
1578
1657
|
for child in root:
|
|
1579
1658
|
if not isinstance(child.tag, str):
|
|
1580
1659
|
continue
|
|
@@ -1594,7 +1673,7 @@ def _get_element_value(
|
|
|
1594
1673
|
|
|
1595
1674
|
def _cached_element_value_factory(
|
|
1596
1675
|
root: _Element,
|
|
1597
|
-
) ->
|
|
1676
|
+
) -> _ElementValueGetter:
|
|
1598
1677
|
"""Create a closure with a child tag index for fast namespace-prefix lookups."""
|
|
1599
1678
|
# Build child tag index once: O(children) instead of O(children × misses)
|
|
1600
1679
|
child_index: dict[str, _Element] = {}
|
|
@@ -1603,7 +1682,9 @@ def _cached_element_value_factory(
|
|
|
1603
1682
|
child_index[child.tag.lower()] = child
|
|
1604
1683
|
|
|
1605
1684
|
def getter(path: str, attribute: Optional[str] = None) -> Optional[str]:
|
|
1606
|
-
return _get_element_value(
|
|
1685
|
+
return _get_element_value(
|
|
1686
|
+
root, path, attribute=attribute, child_index=child_index
|
|
1687
|
+
)
|
|
1607
1688
|
|
|
1608
1689
|
return getter
|
|
1609
1690
|
|
|
@@ -1632,7 +1713,13 @@ def _normalize_iso_datetime_string(value: str) -> str:
|
|
|
1632
1713
|
if cleaned.endswith(("Z", "z")):
|
|
1633
1714
|
cleaned = cleaned[:-1] + "+00:00"
|
|
1634
1715
|
|
|
1635
|
-
if
|
|
1716
|
+
if (
|
|
1717
|
+
" " in cleaned
|
|
1718
|
+
and "T" not in cleaned[:11]
|
|
1719
|
+
and len(cleaned) >= 10
|
|
1720
|
+
and cleaned[4] == "-"
|
|
1721
|
+
and cleaned[0:4].isdigit()
|
|
1722
|
+
):
|
|
1636
1723
|
date_part, rest = cleaned.split(" ", 1)
|
|
1637
1724
|
if rest and rest[0].isdigit():
|
|
1638
1725
|
cleaned = f"{date_part}T{rest}"
|
|
@@ -1688,7 +1775,12 @@ def _fast_rfc822_to_iso(value: str) -> Optional[str]:
|
|
|
1688
1775
|
if tz_offset_seconds == 0:
|
|
1689
1776
|
return f"{base.year:04d}-{base.month:02d}-{base.day:02d}T{h:02d}:{mi:02d}:{s:02d}+00:00"
|
|
1690
1777
|
dt = datetime.datetime(
|
|
1691
|
-
base.year,
|
|
1778
|
+
base.year,
|
|
1779
|
+
base.month,
|
|
1780
|
+
base.day,
|
|
1781
|
+
h,
|
|
1782
|
+
mi,
|
|
1783
|
+
s,
|
|
1692
1784
|
tzinfo=datetime.timezone(datetime.timedelta(seconds=tz_offset_seconds)),
|
|
1693
1785
|
)
|
|
1694
1786
|
utc = dt.astimezone(_UTC)
|
|
@@ -1696,7 +1788,12 @@ def _fast_rfc822_to_iso(value: str) -> Optional[str]:
|
|
|
1696
1788
|
if tz_offset_seconds == 0:
|
|
1697
1789
|
return f"{year}-{month:02d}-{d:02d}T{hour}:{minute}:{second}+00:00"
|
|
1698
1790
|
dt = datetime.datetime(
|
|
1699
|
-
int(year),
|
|
1791
|
+
int(year),
|
|
1792
|
+
month,
|
|
1793
|
+
d,
|
|
1794
|
+
h,
|
|
1795
|
+
mi,
|
|
1796
|
+
s,
|
|
1700
1797
|
tzinfo=datetime.timezone(datetime.timedelta(seconds=tz_offset_seconds)),
|
|
1701
1798
|
)
|
|
1702
1799
|
utc = dt.astimezone(_UTC)
|
|
@@ -1777,7 +1874,9 @@ def _slow_dateparser(value: str) -> Optional[datetime.datetime]:
|
|
|
1777
1874
|
except ImportError:
|
|
1778
1875
|
return None
|
|
1779
1876
|
try:
|
|
1780
|
-
return _dateparser.parse(
|
|
1877
|
+
return _dateparser.parse(
|
|
1878
|
+
value, languages=["en"], settings={**_DATEPARSER_SETTINGS}
|
|
1879
|
+
)
|
|
1781
1880
|
except (ValueError, TypeError):
|
|
1782
1881
|
return None
|
|
1783
1882
|
|
|
@@ -1839,11 +1938,17 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1839
1938
|
base = datetime.date.fromisoformat(m24.group(1))
|
|
1840
1939
|
mins, secs = int(m24.group(2)), int(m24.group(3))
|
|
1841
1940
|
next_day = base + datetime.timedelta(days=1)
|
|
1842
|
-
candidate =
|
|
1941
|
+
candidate = (
|
|
1942
|
+
candidate[: m24.start()]
|
|
1943
|
+
+ f"{next_day}T00:{mins:02d}:{secs:02d}"
|
|
1944
|
+
+ candidate[m24.end() :]
|
|
1945
|
+
)
|
|
1843
1946
|
|
|
1844
1947
|
dt: Optional[datetime.datetime] = None
|
|
1845
1948
|
|
|
1846
|
-
is_iso_like =
|
|
1949
|
+
is_iso_like = (
|
|
1950
|
+
len(candidate) >= 10 and candidate[4] == "-" and candidate[0:4].isdigit()
|
|
1951
|
+
)
|
|
1847
1952
|
if is_iso_like:
|
|
1848
1953
|
iso_candidate = _normalize_iso_datetime_string(candidate)
|
|
1849
1954
|
try:
|
|
@@ -34,12 +34,18 @@ def test_parse_bytes_with_non_utf8_encoding():
|
|
|
34
34
|
|
|
35
35
|
def test_meta_refresh_extraction():
|
|
36
36
|
html = '<!doctype html><html><head><meta http-equiv=refresh content="0; url=https://example.com/feed.xml"></head></html>'
|
|
37
|
-
assert
|
|
37
|
+
assert (
|
|
38
|
+
_extract_meta_refresh_url(html, "https://example.com/feed/")
|
|
39
|
+
== "https://example.com/feed.xml"
|
|
40
|
+
)
|
|
38
41
|
|
|
39
42
|
|
|
40
43
|
def test_meta_refresh_relative_url():
|
|
41
44
|
html = b'<html><head><meta http-equiv="refresh" content="0;url=/index.xml"></head></html>'
|
|
42
|
-
assert
|
|
45
|
+
assert (
|
|
46
|
+
_extract_meta_refresh_url(html, "https://example.com/feed/")
|
|
47
|
+
== "https://example.com/index.xml"
|
|
48
|
+
)
|
|
43
49
|
|
|
44
50
|
|
|
45
51
|
def test_meta_refresh_none_when_missing():
|
|
@@ -50,4 +56,3 @@ def test_meta_refresh_none_when_missing():
|
|
|
50
56
|
def test_meta_refresh_none_when_same_url():
|
|
51
57
|
html = '<html><head><meta http-equiv="refresh" content="0; url=https://example.com/"></head></html>'
|
|
52
58
|
assert _extract_meta_refresh_url(html, "https://example.com/") is None
|
|
53
|
-
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{fastfeedparser-0.5.3 → fastfeedparser-0.5.4}/src/fastfeedparser.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|