fastfeedparser 0.4.8__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fastfeedparser-0.4.8/src/fastfeedparser.egg-info → fastfeedparser-0.5.0}/PKG-INFO +1 -1
- {fastfeedparser-0.4.8 → fastfeedparser-0.5.0}/setup.cfg +1 -1
- {fastfeedparser-0.4.8 → fastfeedparser-0.5.0}/src/fastfeedparser/__init__.py +1 -1
- {fastfeedparser-0.4.8 → fastfeedparser-0.5.0}/src/fastfeedparser/main.py +240 -65
- {fastfeedparser-0.4.8 → fastfeedparser-0.5.0/src/fastfeedparser.egg-info}/PKG-INFO +1 -1
- fastfeedparser-0.5.0/tests/test_encoding.py +53 -0
- fastfeedparser-0.4.8/tests/test_encoding.py +0 -32
- {fastfeedparser-0.4.8 → fastfeedparser-0.5.0}/LICENSE +0 -0
- {fastfeedparser-0.4.8 → fastfeedparser-0.5.0}/README.md +0 -0
- {fastfeedparser-0.4.8 → fastfeedparser-0.5.0}/pyproject.toml +0 -0
- {fastfeedparser-0.4.8 → fastfeedparser-0.5.0}/src/fastfeedparser.egg-info/SOURCES.txt +0 -0
- {fastfeedparser-0.4.8 → fastfeedparser-0.5.0}/src/fastfeedparser.egg-info/dependency_links.txt +0 -0
- {fastfeedparser-0.4.8 → fastfeedparser-0.5.0}/src/fastfeedparser.egg-info/requires.txt +0 -0
- {fastfeedparser-0.4.8 → fastfeedparser-0.5.0}/src/fastfeedparser.egg-info/top_level.txt +0 -0
- {fastfeedparser-0.4.8 → fastfeedparser-0.5.0}/tests/test_integration.py +0 -0
|
@@ -3,6 +3,7 @@ from __future__ import annotations
|
|
|
3
3
|
import datetime
|
|
4
4
|
from email.utils import parsedate_to_datetime
|
|
5
5
|
import gzip
|
|
6
|
+
import html as _html_mod
|
|
6
7
|
import json
|
|
7
8
|
import re
|
|
8
9
|
import zlib
|
|
@@ -15,6 +16,7 @@ try:
|
|
|
15
16
|
except ImportError:
|
|
16
17
|
HAS_BROTLI = False
|
|
17
18
|
from typing import Any, Callable, Optional, TYPE_CHECKING, Literal
|
|
19
|
+
from urllib.parse import urljoin
|
|
18
20
|
from urllib.request import (
|
|
19
21
|
HTTPErrorProcessor,
|
|
20
22
|
HTTPRedirectHandler,
|
|
@@ -58,8 +60,8 @@ _RE_UNCLOSED_LINK_BYTES = re.compile(
|
|
|
58
60
|
br"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
|
|
59
61
|
)
|
|
60
62
|
_RE_FEB29 = re.compile(r"(\d{4})-02-29")
|
|
63
|
+
_RE_HTML_TAGS = re.compile(r"<[^>]+>")
|
|
61
64
|
_RE_WHITESPACE = re.compile(r"\s+")
|
|
62
|
-
_RE_ISO_LIKE = re.compile(r"^\d{4}-\d{2}-\d{2}")
|
|
63
65
|
_RE_ISO_TZ_NO_COLON = re.compile(r"([+-]\d{2})(\d{2})$")
|
|
64
66
|
_RE_ISO_TZ_HOUR_ONLY = re.compile(r"([+-]\d{2})$")
|
|
65
67
|
_RE_ISO_FRACTION = re.compile(r"\.(\d{7,})(?=(?:[+-]\d{2}:?\d{2}|Z|$))", re.IGNORECASE)
|
|
@@ -594,7 +596,31 @@ def _raise_for_non_feed_root(
|
|
|
594
596
|
raise ValueError(
|
|
595
597
|
"Received XML sitemap instead of feed (sitemap is for search engines, not a feed)"
|
|
596
598
|
)
|
|
597
|
-
|
|
599
|
+
|
|
600
|
+
|
|
601
|
+
_RE_META_REFRESH_URL = re.compile(
|
|
602
|
+
r'url\s*=\s*["\']?\s*([^"\'>\s]+)', re.IGNORECASE
|
|
603
|
+
)
|
|
604
|
+
|
|
605
|
+
|
|
606
|
+
def _extract_meta_refresh_url(content: str | bytes, base_url: str) -> str | None:
|
|
607
|
+
"""Extract redirect URL from an HTML meta-refresh tag."""
|
|
608
|
+
html_bytes = content.encode("utf-8") if isinstance(content, str) else content
|
|
609
|
+
try:
|
|
610
|
+
doc = etree.fromstring(html_bytes, parser=etree.HTMLParser())
|
|
611
|
+
except Exception:
|
|
612
|
+
return None
|
|
613
|
+
if doc is None:
|
|
614
|
+
return None
|
|
615
|
+
|
|
616
|
+
for meta in doc.iter("meta"):
|
|
617
|
+
if (meta.get("http-equiv") or "").lower() == "refresh":
|
|
618
|
+
match = _RE_META_REFRESH_URL.search(meta.get("content", ""))
|
|
619
|
+
if match:
|
|
620
|
+
url = urljoin(base_url, match.group(1))
|
|
621
|
+
if url != base_url:
|
|
622
|
+
return url
|
|
623
|
+
return None
|
|
598
624
|
|
|
599
625
|
|
|
600
626
|
def _detect_feed_structure(
|
|
@@ -704,24 +730,26 @@ def _detect_feed_structure(
|
|
|
704
730
|
raise ValueError(f"Unknown feed type: {root.tag}")
|
|
705
731
|
|
|
706
732
|
|
|
707
|
-
def
|
|
708
|
-
"""
|
|
733
|
+
def _should_parse_media_content(root: _Element, xml_content: bytes) -> bool:
|
|
734
|
+
"""Check if feed likely contains Media RSS fields."""
|
|
735
|
+
ns_values = root.nsmap.values() if root.nsmap else ()
|
|
736
|
+
for ns_value in ns_values:
|
|
737
|
+
if not ns_value:
|
|
738
|
+
continue
|
|
739
|
+
if "search.yahoo.com/mrss" in ns_value:
|
|
740
|
+
return True
|
|
709
741
|
|
|
710
|
-
|
|
711
|
-
|
|
742
|
+
# Fallback for feeds with undeclared/late namespace usage.
|
|
743
|
+
return b"search.yahoo.com/mrss" in xml_content or b"<media:" in xml_content
|
|
712
744
|
|
|
713
|
-
Returns:
|
|
714
|
-
FastFeedParserDict containing parsed feed data
|
|
715
745
|
|
|
716
|
-
|
|
717
|
-
|
|
718
|
-
|
|
719
|
-
|
|
720
|
-
if isinstance(source, str) and source.startswith(("http://", "https://")):
|
|
721
|
-
xml_content = _fetch_url_content(source)
|
|
722
|
-
else:
|
|
723
|
-
xml_content = source
|
|
746
|
+
def _should_parse_enclosures(feed_type: _FeedType, xml_content: bytes) -> bool:
|
|
747
|
+
"""Check if feed likely contains RSS enclosure elements."""
|
|
748
|
+
return feed_type == "rss" and b"<enclosure" in xml_content
|
|
749
|
+
|
|
724
750
|
|
|
751
|
+
def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
|
|
752
|
+
"""Parse feed content (XML or JSON) that has already been fetched."""
|
|
725
753
|
json_feed = _maybe_parse_json_feed(xml_content)
|
|
726
754
|
if json_feed is not None:
|
|
727
755
|
return json_feed
|
|
@@ -734,6 +762,8 @@ def parse(source: str | bytes) -> FastFeedParserDict:
|
|
|
734
762
|
feed_type, channel, items, atom_namespace = _detect_feed_structure(
|
|
735
763
|
root, xml_content, root_tag_local
|
|
736
764
|
)
|
|
765
|
+
parse_media_content = _should_parse_media_content(root, xml_content)
|
|
766
|
+
parse_enclosures = _should_parse_enclosures(feed_type, xml_content)
|
|
737
767
|
|
|
738
768
|
feed = _parse_feed_info(channel, feed_type, atom_namespace)
|
|
739
769
|
|
|
@@ -741,7 +771,13 @@ def parse(source: str | bytes) -> FastFeedParserDict:
|
|
|
741
771
|
entries: list[FastFeedParserDict] = []
|
|
742
772
|
feed["entries"] = entries
|
|
743
773
|
for item in items:
|
|
744
|
-
entry = _parse_feed_entry(
|
|
774
|
+
entry = _parse_feed_entry(
|
|
775
|
+
item,
|
|
776
|
+
feed_type,
|
|
777
|
+
atom_namespace,
|
|
778
|
+
parse_media_content=parse_media_content,
|
|
779
|
+
parse_enclosures=parse_enclosures,
|
|
780
|
+
)
|
|
745
781
|
# Ensure that titles and descriptions are always present
|
|
746
782
|
entry["title"] = entry.get("title", "").strip()
|
|
747
783
|
entry["description"] = entry.get("description", "").strip()
|
|
@@ -750,6 +786,39 @@ def parse(source: str | bytes) -> FastFeedParserDict:
|
|
|
750
786
|
return feed
|
|
751
787
|
|
|
752
788
|
|
|
789
|
+
def parse(source: str | bytes) -> FastFeedParserDict:
|
|
790
|
+
"""Parse a feed from a URL or XML content.
|
|
791
|
+
|
|
792
|
+
Args:
|
|
793
|
+
source: URL string or XML content string/bytes
|
|
794
|
+
|
|
795
|
+
Returns:
|
|
796
|
+
FastFeedParserDict containing parsed feed data
|
|
797
|
+
|
|
798
|
+
Raises:
|
|
799
|
+
ValueError: If content is empty or invalid
|
|
800
|
+
HTTPError: If URL fetch fails
|
|
801
|
+
"""
|
|
802
|
+
is_url = isinstance(source, str) and source.startswith(("http://", "https://"))
|
|
803
|
+
if is_url:
|
|
804
|
+
content = _fetch_url_content(source)
|
|
805
|
+
else:
|
|
806
|
+
content = source
|
|
807
|
+
|
|
808
|
+
try:
|
|
809
|
+
return _parse_content(content)
|
|
810
|
+
except ValueError as e:
|
|
811
|
+
if not is_url:
|
|
812
|
+
raise
|
|
813
|
+
err_msg = str(e)
|
|
814
|
+
if "HTML" not in err_msg and "not a valid RSS/Atom feed" not in err_msg:
|
|
815
|
+
raise
|
|
816
|
+
redirect_url = _extract_meta_refresh_url(content, source)
|
|
817
|
+
if redirect_url is None:
|
|
818
|
+
raise
|
|
819
|
+
return parse(redirect_url)
|
|
820
|
+
|
|
821
|
+
|
|
753
822
|
def _parse_feed_info(
|
|
754
823
|
channel: _Element, feed_type: _FeedType, atom_namespace: Optional[str] = None
|
|
755
824
|
) -> FastFeedParserDict:
|
|
@@ -1054,16 +1123,9 @@ def _populate_entry_content(
|
|
|
1054
1123
|
content_value = entry["content"][0]["value"]
|
|
1055
1124
|
if content_value:
|
|
1056
1125
|
if "<" in content_value:
|
|
1057
|
-
|
|
1058
|
-
|
|
1059
|
-
|
|
1060
|
-
content_text = html_content.xpath("string()")
|
|
1061
|
-
if isinstance(content_text, str):
|
|
1062
|
-
content_value = _RE_WHITESPACE.sub(" ", content_text)
|
|
1063
|
-
except etree.ParserError:
|
|
1064
|
-
pass
|
|
1065
|
-
else:
|
|
1066
|
-
content_value = _RE_WHITESPACE.sub(" ", content_value)
|
|
1126
|
+
content_value = _RE_HTML_TAGS.sub(" ", content_value[:2048])
|
|
1127
|
+
content_value = _html_mod.unescape(content_value)
|
|
1128
|
+
content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
|
|
1067
1129
|
entry["description"] = content_value[:512]
|
|
1068
1130
|
|
|
1069
1131
|
|
|
@@ -1154,12 +1216,133 @@ def _parse_enclosures(item: _Element) -> list[dict[str, Any]] | None:
|
|
|
1154
1216
|
return enclosures or None
|
|
1155
1217
|
|
|
1156
1218
|
|
|
1219
|
+
def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]], dict[str, Optional[str]]]:
|
|
1220
|
+
by_local: dict[str, Optional[str]] = {}
|
|
1221
|
+
by_full: dict[str, Optional[str]] = {}
|
|
1222
|
+
for child in item:
|
|
1223
|
+
tag = child.tag
|
|
1224
|
+
if not isinstance(tag, str):
|
|
1225
|
+
continue
|
|
1226
|
+
text_value = child.text.strip() if child.text else None
|
|
1227
|
+
if tag not in by_full:
|
|
1228
|
+
by_full[tag] = text_value
|
|
1229
|
+
local = tag.rsplit("}", 1)[-1].lower()
|
|
1230
|
+
if ":" in local:
|
|
1231
|
+
local = local.split(":", 1)[1]
|
|
1232
|
+
if local not in by_local:
|
|
1233
|
+
by_local[local] = text_value
|
|
1234
|
+
return by_local, by_full
|
|
1235
|
+
|
|
1236
|
+
|
|
1237
|
+
def _first_non_empty(mapping: dict[str, Optional[str]], keys: tuple[str, ...]) -> Optional[str]:
|
|
1238
|
+
for key in keys:
|
|
1239
|
+
value = mapping.get(key)
|
|
1240
|
+
if value:
|
|
1241
|
+
return value
|
|
1242
|
+
return None
|
|
1243
|
+
|
|
1244
|
+
|
|
1245
|
+
def _parse_rss_feed_entry_fast(
|
|
1246
|
+
item: _Element,
|
|
1247
|
+
atom_ns: str,
|
|
1248
|
+
parse_media_content: bool = True,
|
|
1249
|
+
parse_enclosures: bool = True,
|
|
1250
|
+
) -> FastFeedParserDict:
|
|
1251
|
+
text_by_local, text_by_full = _build_rss_item_text_maps(item)
|
|
1252
|
+
|
|
1253
|
+
entry = FastFeedParserDict()
|
|
1254
|
+
atom_id = text_by_full.get(f"{{{atom_ns}}}id")
|
|
1255
|
+
rss_guid = text_by_local.get("guid")
|
|
1256
|
+
rdf_about = item.get("{http://www.w3.org/1999/02/22-rdf-syntax-ns#}about")
|
|
1257
|
+
entry_id: Optional[str] = atom_id or rss_guid or rdf_about
|
|
1258
|
+
if entry_id:
|
|
1259
|
+
entry["id"] = entry_id.strip()
|
|
1260
|
+
|
|
1261
|
+
title = text_by_local.get("title")
|
|
1262
|
+
if title:
|
|
1263
|
+
entry["title"] = title
|
|
1264
|
+
|
|
1265
|
+
description = _first_non_empty(text_by_local, ("description", "summary"))
|
|
1266
|
+
if description:
|
|
1267
|
+
entry["description"] = description
|
|
1268
|
+
|
|
1269
|
+
link = text_by_local.get("link")
|
|
1270
|
+
if link:
|
|
1271
|
+
entry["link"] = link
|
|
1272
|
+
|
|
1273
|
+
published_source = _first_non_empty(text_by_local, ("pubdate", "published", "issued", "date"))
|
|
1274
|
+
if published_source:
|
|
1275
|
+
published = _parse_date(published_source)
|
|
1276
|
+
if published:
|
|
1277
|
+
entry["published"] = published
|
|
1278
|
+
|
|
1279
|
+
updated_source = _first_non_empty(text_by_local, ("lastbuilddate", "updated", "modified"))
|
|
1280
|
+
if updated_source:
|
|
1281
|
+
updated = _parse_date(updated_source)
|
|
1282
|
+
if updated:
|
|
1283
|
+
entry["updated"] = updated
|
|
1284
|
+
|
|
1285
|
+
if "published" not in entry and rss_guid:
|
|
1286
|
+
guid_date = _parse_date(rss_guid)
|
|
1287
|
+
if guid_date:
|
|
1288
|
+
entry["published"] = guid_date
|
|
1289
|
+
|
|
1290
|
+
if "updated" in entry and "published" not in entry:
|
|
1291
|
+
entry["published"] = entry["updated"]
|
|
1292
|
+
|
|
1293
|
+
_populate_entry_links(entry, item, atom_ns)
|
|
1294
|
+
if "id" not in entry and "link" in entry:
|
|
1295
|
+
entry["id"] = entry["link"]
|
|
1296
|
+
|
|
1297
|
+
_populate_entry_content(entry, item, "rss", atom_ns)
|
|
1298
|
+
|
|
1299
|
+
if parse_media_content:
|
|
1300
|
+
media_contents = _parse_media_content(item)
|
|
1301
|
+
if media_contents:
|
|
1302
|
+
entry["media_content"] = media_contents
|
|
1303
|
+
|
|
1304
|
+
if parse_enclosures:
|
|
1305
|
+
enclosures = _parse_enclosures(item)
|
|
1306
|
+
if enclosures:
|
|
1307
|
+
entry["enclosures"] = enclosures
|
|
1308
|
+
|
|
1309
|
+
author = _first_non_empty(text_by_local, ("author", "creator"))
|
|
1310
|
+
if not author:
|
|
1311
|
+
atom_author = item.find(f"{{{atom_ns}}}author/{{{atom_ns}}}name")
|
|
1312
|
+
author = atom_author.text.strip() if atom_author is not None and atom_author.text else None
|
|
1313
|
+
if author:
|
|
1314
|
+
entry["author"] = author
|
|
1315
|
+
|
|
1316
|
+
comments = text_by_local.get("comments")
|
|
1317
|
+
if comments:
|
|
1318
|
+
entry["comments"] = comments
|
|
1319
|
+
|
|
1320
|
+
tags = _parse_tags(item, "rss", atom_ns)
|
|
1321
|
+
if tags:
|
|
1322
|
+
entry["tags"] = tags
|
|
1323
|
+
|
|
1324
|
+
return entry
|
|
1325
|
+
|
|
1326
|
+
|
|
1157
1327
|
def _parse_feed_entry(
|
|
1158
|
-
item: _Element,
|
|
1328
|
+
item: _Element,
|
|
1329
|
+
feed_type: _FeedType,
|
|
1330
|
+
atom_namespace: Optional[str] = None,
|
|
1331
|
+
*,
|
|
1332
|
+
parse_media_content: bool = True,
|
|
1333
|
+
parse_enclosures: bool = True,
|
|
1159
1334
|
) -> FastFeedParserDict:
|
|
1160
1335
|
# Use dynamic atom namespace or fallback to default
|
|
1161
1336
|
atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
|
|
1162
1337
|
|
|
1338
|
+
if feed_type == "rss":
|
|
1339
|
+
return _parse_rss_feed_entry_fast(
|
|
1340
|
+
item,
|
|
1341
|
+
atom_ns,
|
|
1342
|
+
parse_media_content=parse_media_content,
|
|
1343
|
+
parse_enclosures=parse_enclosures,
|
|
1344
|
+
)
|
|
1345
|
+
|
|
1163
1346
|
# Check if this is Atom 0.3 to use different date field names
|
|
1164
1347
|
is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
|
|
1165
1348
|
|
|
@@ -1272,38 +1455,27 @@ def _parse_feed_entry(
|
|
|
1272
1455
|
|
|
1273
1456
|
_populate_entry_content(entry, item, feed_type, atom_ns)
|
|
1274
1457
|
|
|
1275
|
-
|
|
1276
|
-
|
|
1277
|
-
|
|
1278
|
-
|
|
1279
|
-
|
|
1280
|
-
if
|
|
1281
|
-
|
|
1282
|
-
|
|
1283
|
-
|
|
1284
|
-
|
|
1285
|
-
|
|
1286
|
-
|
|
1287
|
-
|
|
1288
|
-
|
|
1289
|
-
|
|
1290
|
-
or get_field_value(
|
|
1291
|
-
"{http://purl.org/dc/elements/1.1/}creator",
|
|
1292
|
-
"{http://purl.org/dc/elements/1.1/}creator",
|
|
1293
|
-
"{http://purl.org/dc/elements/1.1/}creator",
|
|
1294
|
-
False,
|
|
1295
|
-
)
|
|
1296
|
-
or element_get("{http://purl.org/dc/elements/1.1/}creator")
|
|
1297
|
-
or element_get("author")
|
|
1458
|
+
if parse_media_content:
|
|
1459
|
+
media_contents = _parse_media_content(item)
|
|
1460
|
+
if media_contents:
|
|
1461
|
+
entry["media_content"] = media_contents
|
|
1462
|
+
|
|
1463
|
+
if parse_enclosures:
|
|
1464
|
+
enclosures = _parse_enclosures(item)
|
|
1465
|
+
if enclosures:
|
|
1466
|
+
entry["enclosures"] = enclosures
|
|
1467
|
+
|
|
1468
|
+
author = get_field_value(
|
|
1469
|
+
"author",
|
|
1470
|
+
f"{{{atom_ns}}}author/{{{atom_ns}}}name",
|
|
1471
|
+
"{http://purl.org/dc/elements/1.1/}creator",
|
|
1472
|
+
False,
|
|
1298
1473
|
)
|
|
1474
|
+
if not author:
|
|
1475
|
+
author = element_get("{http://purl.org/dc/elements/1.1/}creator") or element_get("author")
|
|
1299
1476
|
if author:
|
|
1300
1477
|
entry["author"] = author
|
|
1301
1478
|
|
|
1302
|
-
if feed_type == "rss":
|
|
1303
|
-
comments = element_get("comments")
|
|
1304
|
-
if comments:
|
|
1305
|
-
entry["comments"] = comments
|
|
1306
|
-
|
|
1307
1479
|
# Parse entry-level tags/categories
|
|
1308
1480
|
tags = _parse_tags(item, feed_type, atom_ns)
|
|
1309
1481
|
if tags:
|
|
@@ -1456,7 +1628,7 @@ def _normalize_iso_datetime_string(value: str) -> str:
|
|
|
1456
1628
|
if cleaned.endswith(("Z", "z")):
|
|
1457
1629
|
cleaned = cleaned[:-1] + "+00:00"
|
|
1458
1630
|
|
|
1459
|
-
if " " in cleaned and "T" not in cleaned[:11] and
|
|
1631
|
+
if " " in cleaned and "T" not in cleaned[:11] and len(cleaned) >= 10 and cleaned[4] == "-" and cleaned[0:4].isdigit():
|
|
1460
1632
|
date_part, rest = cleaned.split(" ", 1)
|
|
1461
1633
|
if rest and rest[0].isdigit():
|
|
1462
1634
|
cleaned = f"{date_part}T{rest}"
|
|
@@ -1572,18 +1744,20 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1572
1744
|
if not date_str:
|
|
1573
1745
|
return None
|
|
1574
1746
|
|
|
1575
|
-
candidate =
|
|
1747
|
+
candidate = date_str.strip()
|
|
1576
1748
|
if not candidate:
|
|
1577
1749
|
return None
|
|
1750
|
+
if "\n" in candidate or "\r" in candidate or "\t" in candidate or " " in candidate:
|
|
1751
|
+
candidate = _RE_WHITESPACE.sub(" ", candidate)
|
|
1578
1752
|
|
|
1579
1753
|
# Fix invalid leap year dates (Feb 29 in non-leap years)
|
|
1580
1754
|
# This handles feeds with incorrect dates like "2023-02-29"
|
|
1581
|
-
|
|
1582
|
-
|
|
1583
|
-
|
|
1584
|
-
|
|
1585
|
-
|
|
1586
|
-
|
|
1755
|
+
if "-02-29" in candidate:
|
|
1756
|
+
year_match = _RE_FEB29.match(candidate)
|
|
1757
|
+
if year_match:
|
|
1758
|
+
year = int(year_match.group(1))
|
|
1759
|
+
if not ((year % 4 == 0 and year % 100 != 0) or (year % 400 == 0)):
|
|
1760
|
+
candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
|
|
1587
1761
|
|
|
1588
1762
|
if "24:00" in candidate:
|
|
1589
1763
|
candidate = candidate.replace("24:00:00", "00:00:00").replace(
|
|
@@ -1592,7 +1766,8 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1592
1766
|
|
|
1593
1767
|
dt: Optional[datetime.datetime] = None
|
|
1594
1768
|
|
|
1595
|
-
|
|
1769
|
+
is_iso_like = len(candidate) >= 10 and candidate[4] == "-" and candidate[0:4].isdigit()
|
|
1770
|
+
if is_iso_like:
|
|
1596
1771
|
iso_candidate = _normalize_iso_datetime_string(candidate)
|
|
1597
1772
|
try:
|
|
1598
1773
|
dt = datetime.datetime.fromisoformat(iso_candidate)
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
from fastfeedparser import parse
|
|
2
|
+
from fastfeedparser.main import _extract_meta_refresh_url
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def test_parse_str_with_non_utf8_xml_declaration():
|
|
6
|
+
xml = (
|
|
7
|
+
'<?xml version="1.0" encoding="iso-8859-1"?>'
|
|
8
|
+
'<rss version="2.0">'
|
|
9
|
+
"<channel>"
|
|
10
|
+
"<title>café</title>"
|
|
11
|
+
"<item><title>café</title></item>"
|
|
12
|
+
"</channel>"
|
|
13
|
+
"</rss>"
|
|
14
|
+
)
|
|
15
|
+
feed = parse(xml)
|
|
16
|
+
assert feed.feed.title == "café"
|
|
17
|
+
assert feed.entries[0].title == "café"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def test_parse_bytes_with_non_utf8_encoding():
|
|
21
|
+
xml_bytes = (
|
|
22
|
+
b'<?xml version="1.0" encoding="iso-8859-1"?>'
|
|
23
|
+
b'<rss version="2.0">'
|
|
24
|
+
b"<channel>"
|
|
25
|
+
b"<title>caf\xe9</title>"
|
|
26
|
+
b"<item><title>caf\xe9</title></item>"
|
|
27
|
+
b"</channel>"
|
|
28
|
+
b"</rss>"
|
|
29
|
+
)
|
|
30
|
+
feed = parse(xml_bytes)
|
|
31
|
+
assert feed.feed.title == "café"
|
|
32
|
+
assert feed.entries[0].title == "café"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def test_meta_refresh_extraction():
|
|
36
|
+
html = '<!doctype html><html><head><meta http-equiv=refresh content="0; url=https://example.com/feed.xml"></head></html>'
|
|
37
|
+
assert _extract_meta_refresh_url(html, "https://example.com/feed/") == "https://example.com/feed.xml"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def test_meta_refresh_relative_url():
|
|
41
|
+
html = b'<html><head><meta http-equiv="refresh" content="0;url=/index.xml"></head></html>'
|
|
42
|
+
assert _extract_meta_refresh_url(html, "https://example.com/feed/") == "https://example.com/index.xml"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def test_meta_refresh_none_when_missing():
|
|
46
|
+
html = "<html><head><title>Hello</title></head><body></body></html>"
|
|
47
|
+
assert _extract_meta_refresh_url(html, "https://example.com/") is None
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def test_meta_refresh_none_when_same_url():
|
|
51
|
+
html = '<html><head><meta http-equiv="refresh" content="0; url=https://example.com/"></head></html>'
|
|
52
|
+
assert _extract_meta_refresh_url(html, "https://example.com/") is None
|
|
53
|
+
|
|
@@ -1,32 +0,0 @@
|
|
|
1
|
-
from fastfeedparser import parse
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
def test_parse_str_with_non_utf8_xml_declaration():
|
|
5
|
-
xml = (
|
|
6
|
-
'<?xml version="1.0" encoding="iso-8859-1"?>'
|
|
7
|
-
'<rss version="2.0">'
|
|
8
|
-
"<channel>"
|
|
9
|
-
"<title>café</title>"
|
|
10
|
-
"<item><title>café</title></item>"
|
|
11
|
-
"</channel>"
|
|
12
|
-
"</rss>"
|
|
13
|
-
)
|
|
14
|
-
feed = parse(xml)
|
|
15
|
-
assert feed.feed.title == "café"
|
|
16
|
-
assert feed.entries[0].title == "café"
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
def test_parse_bytes_with_non_utf8_encoding():
|
|
20
|
-
xml_bytes = (
|
|
21
|
-
b'<?xml version="1.0" encoding="iso-8859-1"?>'
|
|
22
|
-
b'<rss version="2.0">'
|
|
23
|
-
b"<channel>"
|
|
24
|
-
b"<title>caf\xe9</title>"
|
|
25
|
-
b"<item><title>caf\xe9</title></item>"
|
|
26
|
-
b"</channel>"
|
|
27
|
-
b"</rss>"
|
|
28
|
-
)
|
|
29
|
-
feed = parse(xml_bytes)
|
|
30
|
-
assert feed.feed.title == "café"
|
|
31
|
-
assert feed.entries[0].title == "café"
|
|
32
|
-
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{fastfeedparser-0.4.8 → fastfeedparser-0.5.0}/src/fastfeedparser.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|