fastfeedparser 0.4.9__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fastfeedparser-0.4.9/src/fastfeedparser.egg-info → fastfeedparser-0.5.0}/PKG-INFO +1 -1
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.0}/setup.cfg +1 -1
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.0}/src/fastfeedparser/__init__.py +1 -1
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.0}/src/fastfeedparser/main.py +23 -43
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.0/src/fastfeedparser.egg-info}/PKG-INFO +1 -1
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.0}/LICENSE +0 -0
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.0}/README.md +0 -0
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.0}/pyproject.toml +0 -0
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.0}/src/fastfeedparser.egg-info/SOURCES.txt +0 -0
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.0}/src/fastfeedparser.egg-info/dependency_links.txt +0 -0
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.0}/src/fastfeedparser.egg-info/requires.txt +0 -0
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.0}/src/fastfeedparser.egg-info/top_level.txt +0 -0
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.0}/tests/test_encoding.py +0 -0
- {fastfeedparser-0.4.9 → fastfeedparser-0.5.0}/tests/test_integration.py +0 -0
|
@@ -3,6 +3,7 @@ from __future__ import annotations
|
|
|
3
3
|
import datetime
|
|
4
4
|
from email.utils import parsedate_to_datetime
|
|
5
5
|
import gzip
|
|
6
|
+
import html as _html_mod
|
|
6
7
|
import json
|
|
7
8
|
import re
|
|
8
9
|
import zlib
|
|
@@ -59,8 +60,8 @@ _RE_UNCLOSED_LINK_BYTES = re.compile(
|
|
|
59
60
|
br"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
|
|
60
61
|
)
|
|
61
62
|
_RE_FEB29 = re.compile(r"(\d{4})-02-29")
|
|
63
|
+
_RE_HTML_TAGS = re.compile(r"<[^>]+>")
|
|
62
64
|
_RE_WHITESPACE = re.compile(r"\s+")
|
|
63
|
-
_RE_ISO_LIKE = re.compile(r"^\d{4}-\d{2}-\d{2}")
|
|
64
65
|
_RE_ISO_TZ_NO_COLON = re.compile(r"([+-]\d{2})(\d{2})$")
|
|
65
66
|
_RE_ISO_TZ_HOUR_ONLY = re.compile(r"([+-]\d{2})$")
|
|
66
67
|
_RE_ISO_FRACTION = re.compile(r"\.(\d{7,})(?=(?:[+-]\d{2}:?\d{2}|Z|$))", re.IGNORECASE)
|
|
@@ -595,7 +596,6 @@ def _raise_for_non_feed_root(
|
|
|
595
596
|
raise ValueError(
|
|
596
597
|
"Received XML sitemap instead of feed (sitemap is for search engines, not a feed)"
|
|
597
598
|
)
|
|
598
|
-
raise ValueError(f"Not a valid feed: {root_tag_local} element found - {error_msg[:100]}")
|
|
599
599
|
|
|
600
600
|
|
|
601
601
|
_RE_META_REFRESH_URL = re.compile(
|
|
@@ -1123,16 +1123,9 @@ def _populate_entry_content(
|
|
|
1123
1123
|
content_value = entry["content"][0]["value"]
|
|
1124
1124
|
if content_value:
|
|
1125
1125
|
if "<" in content_value:
|
|
1126
|
-
|
|
1127
|
-
|
|
1128
|
-
|
|
1129
|
-
content_text = html_content.xpath("string()")
|
|
1130
|
-
if isinstance(content_text, str):
|
|
1131
|
-
content_value = _RE_WHITESPACE.sub(" ", content_text)
|
|
1132
|
-
except etree.ParserError:
|
|
1133
|
-
pass
|
|
1134
|
-
else:
|
|
1135
|
-
content_value = _RE_WHITESPACE.sub(" ", content_value)
|
|
1126
|
+
content_value = _RE_HTML_TAGS.sub(" ", content_value[:2048])
|
|
1127
|
+
content_value = _html_mod.unescape(content_value)
|
|
1128
|
+
content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
|
|
1136
1129
|
entry["description"] = content_value[:512]
|
|
1137
1130
|
|
|
1138
1131
|
|
|
@@ -1223,13 +1216,6 @@ def _parse_enclosures(item: _Element) -> list[dict[str, Any]] | None:
|
|
|
1223
1216
|
return enclosures or None
|
|
1224
1217
|
|
|
1225
1218
|
|
|
1226
|
-
def _normalize_local_tag_name(tag: str) -> str:
|
|
1227
|
-
local = tag.rsplit("}", 1)[-1].lower()
|
|
1228
|
-
if ":" in local:
|
|
1229
|
-
local = local.split(":", 1)[1]
|
|
1230
|
-
return local
|
|
1231
|
-
|
|
1232
|
-
|
|
1233
1219
|
def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]], dict[str, Optional[str]]]:
|
|
1234
1220
|
by_local: dict[str, Optional[str]] = {}
|
|
1235
1221
|
by_full: dict[str, Optional[str]] = {}
|
|
@@ -1240,7 +1226,9 @@ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]],
|
|
|
1240
1226
|
text_value = child.text.strip() if child.text else None
|
|
1241
1227
|
if tag not in by_full:
|
|
1242
1228
|
by_full[tag] = text_value
|
|
1243
|
-
local =
|
|
1229
|
+
local = tag.rsplit("}", 1)[-1].lower()
|
|
1230
|
+
if ":" in local:
|
|
1231
|
+
local = local.split(":", 1)[1]
|
|
1244
1232
|
if local not in by_local:
|
|
1245
1233
|
by_local[local] = text_value
|
|
1246
1234
|
return by_local, by_full
|
|
@@ -1477,22 +1465,14 @@ def _parse_feed_entry(
|
|
|
1477
1465
|
if enclosures:
|
|
1478
1466
|
entry["enclosures"] = enclosures
|
|
1479
1467
|
|
|
1480
|
-
author = (
|
|
1481
|
-
|
|
1482
|
-
|
|
1483
|
-
|
|
1484
|
-
|
|
1485
|
-
False,
|
|
1486
|
-
)
|
|
1487
|
-
or get_field_value(
|
|
1488
|
-
"{http://purl.org/dc/elements/1.1/}creator",
|
|
1489
|
-
"{http://purl.org/dc/elements/1.1/}creator",
|
|
1490
|
-
"{http://purl.org/dc/elements/1.1/}creator",
|
|
1491
|
-
False,
|
|
1492
|
-
)
|
|
1493
|
-
or element_get("{http://purl.org/dc/elements/1.1/}creator")
|
|
1494
|
-
or element_get("author")
|
|
1468
|
+
author = get_field_value(
|
|
1469
|
+
"author",
|
|
1470
|
+
f"{{{atom_ns}}}author/{{{atom_ns}}}name",
|
|
1471
|
+
"{http://purl.org/dc/elements/1.1/}creator",
|
|
1472
|
+
False,
|
|
1495
1473
|
)
|
|
1474
|
+
if not author:
|
|
1475
|
+
author = element_get("{http://purl.org/dc/elements/1.1/}creator") or element_get("author")
|
|
1496
1476
|
if author:
|
|
1497
1477
|
entry["author"] = author
|
|
1498
1478
|
|
|
@@ -1648,7 +1628,7 @@ def _normalize_iso_datetime_string(value: str) -> str:
|
|
|
1648
1628
|
if cleaned.endswith(("Z", "z")):
|
|
1649
1629
|
cleaned = cleaned[:-1] + "+00:00"
|
|
1650
1630
|
|
|
1651
|
-
if " " in cleaned and "T" not in cleaned[:11] and
|
|
1631
|
+
if " " in cleaned and "T" not in cleaned[:11] and len(cleaned) >= 10 and cleaned[4] == "-" and cleaned[0:4].isdigit():
|
|
1652
1632
|
date_part, rest = cleaned.split(" ", 1)
|
|
1653
1633
|
if rest and rest[0].isdigit():
|
|
1654
1634
|
cleaned = f"{date_part}T{rest}"
|
|
@@ -1772,12 +1752,12 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1772
1752
|
|
|
1773
1753
|
# Fix invalid leap year dates (Feb 29 in non-leap years)
|
|
1774
1754
|
# This handles feeds with incorrect dates like "2023-02-29"
|
|
1775
|
-
|
|
1776
|
-
|
|
1777
|
-
|
|
1778
|
-
|
|
1779
|
-
|
|
1780
|
-
|
|
1755
|
+
if "-02-29" in candidate:
|
|
1756
|
+
year_match = _RE_FEB29.match(candidate)
|
|
1757
|
+
if year_match:
|
|
1758
|
+
year = int(year_match.group(1))
|
|
1759
|
+
if not ((year % 4 == 0 and year % 100 != 0) or (year % 400 == 0)):
|
|
1760
|
+
candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
|
|
1781
1761
|
|
|
1782
1762
|
if "24:00" in candidate:
|
|
1783
1763
|
candidate = candidate.replace("24:00:00", "00:00:00").replace(
|
|
@@ -1786,7 +1766,7 @@ def _parse_date(date_str: str) -> Optional[str]:
|
|
|
1786
1766
|
|
|
1787
1767
|
dt: Optional[datetime.datetime] = None
|
|
1788
1768
|
|
|
1789
|
-
is_iso_like =
|
|
1769
|
+
is_iso_like = len(candidate) >= 10 and candidate[4] == "-" and candidate[0:4].isdigit()
|
|
1790
1770
|
if is_iso_like:
|
|
1791
1771
|
iso_candidate = _normalize_iso_datetime_string(candidate)
|
|
1792
1772
|
try:
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{fastfeedparser-0.4.9 → fastfeedparser-0.5.0}/src/fastfeedparser.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|