fastfeedparser 0.4.9__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.4.9
3
+ Version: 0.5.0
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = fastfeedparser
3
- version = 0.4.9
3
+ version = 0.5.0
4
4
  author = Vladimir Prelovac
5
5
  author_email = vlad@kagi.com
6
6
  description = High performance RSS, Atom, JSON and RDF feed parser in Python
@@ -1,4 +1,4 @@
1
1
  from .main import parse, FastFeedParserDict
2
2
 
3
- __version__ = "0.4.9"
3
+ __version__ = "0.5.0"
4
4
  __all__ = ["parse", "FastFeedParserDict"]
@@ -3,6 +3,7 @@ from __future__ import annotations
3
3
  import datetime
4
4
  from email.utils import parsedate_to_datetime
5
5
  import gzip
6
+ import html as _html_mod
6
7
  import json
7
8
  import re
8
9
  import zlib
@@ -59,8 +60,8 @@ _RE_UNCLOSED_LINK_BYTES = re.compile(
59
60
  br"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
60
61
  )
61
62
  _RE_FEB29 = re.compile(r"(\d{4})-02-29")
63
+ _RE_HTML_TAGS = re.compile(r"<[^>]+>")
62
64
  _RE_WHITESPACE = re.compile(r"\s+")
63
- _RE_ISO_LIKE = re.compile(r"^\d{4}-\d{2}-\d{2}")
64
65
  _RE_ISO_TZ_NO_COLON = re.compile(r"([+-]\d{2})(\d{2})$")
65
66
  _RE_ISO_TZ_HOUR_ONLY = re.compile(r"([+-]\d{2})$")
66
67
  _RE_ISO_FRACTION = re.compile(r"\.(\d{7,})(?=(?:[+-]\d{2}:?\d{2}|Z|$))", re.IGNORECASE)
@@ -595,7 +596,6 @@ def _raise_for_non_feed_root(
595
596
  raise ValueError(
596
597
  "Received XML sitemap instead of feed (sitemap is for search engines, not a feed)"
597
598
  )
598
- raise ValueError(f"Not a valid feed: {root_tag_local} element found - {error_msg[:100]}")
599
599
 
600
600
 
601
601
  _RE_META_REFRESH_URL = re.compile(
@@ -1123,16 +1123,9 @@ def _populate_entry_content(
1123
1123
  content_value = entry["content"][0]["value"]
1124
1124
  if content_value:
1125
1125
  if "<" in content_value:
1126
- try:
1127
- html_content = etree.HTML(content_value)
1128
- if html_content is not None:
1129
- content_text = html_content.xpath("string()")
1130
- if isinstance(content_text, str):
1131
- content_value = _RE_WHITESPACE.sub(" ", content_text)
1132
- except etree.ParserError:
1133
- pass
1134
- else:
1135
- content_value = _RE_WHITESPACE.sub(" ", content_value)
1126
+ content_value = _RE_HTML_TAGS.sub(" ", content_value[:2048])
1127
+ content_value = _html_mod.unescape(content_value)
1128
+ content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
1136
1129
  entry["description"] = content_value[:512]
1137
1130
 
1138
1131
 
@@ -1223,13 +1216,6 @@ def _parse_enclosures(item: _Element) -> list[dict[str, Any]] | None:
1223
1216
  return enclosures or None
1224
1217
 
1225
1218
 
1226
- def _normalize_local_tag_name(tag: str) -> str:
1227
- local = tag.rsplit("}", 1)[-1].lower()
1228
- if ":" in local:
1229
- local = local.split(":", 1)[1]
1230
- return local
1231
-
1232
-
1233
1219
  def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]], dict[str, Optional[str]]]:
1234
1220
  by_local: dict[str, Optional[str]] = {}
1235
1221
  by_full: dict[str, Optional[str]] = {}
@@ -1240,7 +1226,9 @@ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]],
1240
1226
  text_value = child.text.strip() if child.text else None
1241
1227
  if tag not in by_full:
1242
1228
  by_full[tag] = text_value
1243
- local = _normalize_local_tag_name(tag)
1229
+ local = tag.rsplit("}", 1)[-1].lower()
1230
+ if ":" in local:
1231
+ local = local.split(":", 1)[1]
1244
1232
  if local not in by_local:
1245
1233
  by_local[local] = text_value
1246
1234
  return by_local, by_full
@@ -1477,22 +1465,14 @@ def _parse_feed_entry(
1477
1465
  if enclosures:
1478
1466
  entry["enclosures"] = enclosures
1479
1467
 
1480
- author = (
1481
- get_field_value(
1482
- "author",
1483
- f"{{{atom_ns}}}author/{{{atom_ns}}}name",
1484
- "{http://purl.org/dc/elements/1.1/}creator",
1485
- False,
1486
- )
1487
- or get_field_value(
1488
- "{http://purl.org/dc/elements/1.1/}creator",
1489
- "{http://purl.org/dc/elements/1.1/}creator",
1490
- "{http://purl.org/dc/elements/1.1/}creator",
1491
- False,
1492
- )
1493
- or element_get("{http://purl.org/dc/elements/1.1/}creator")
1494
- or element_get("author")
1468
+ author = get_field_value(
1469
+ "author",
1470
+ f"{{{atom_ns}}}author/{{{atom_ns}}}name",
1471
+ "{http://purl.org/dc/elements/1.1/}creator",
1472
+ False,
1495
1473
  )
1474
+ if not author:
1475
+ author = element_get("{http://purl.org/dc/elements/1.1/}creator") or element_get("author")
1496
1476
  if author:
1497
1477
  entry["author"] = author
1498
1478
 
@@ -1648,7 +1628,7 @@ def _normalize_iso_datetime_string(value: str) -> str:
1648
1628
  if cleaned.endswith(("Z", "z")):
1649
1629
  cleaned = cleaned[:-1] + "+00:00"
1650
1630
 
1651
- if " " in cleaned and "T" not in cleaned[:11] and _RE_ISO_LIKE.match(cleaned):
1631
+ if " " in cleaned and "T" not in cleaned[:11] and len(cleaned) >= 10 and cleaned[4] == "-" and cleaned[0:4].isdigit():
1652
1632
  date_part, rest = cleaned.split(" ", 1)
1653
1633
  if rest and rest[0].isdigit():
1654
1634
  cleaned = f"{date_part}T{rest}"
@@ -1772,12 +1752,12 @@ def _parse_date(date_str: str) -> Optional[str]:
1772
1752
 
1773
1753
  # Fix invalid leap year dates (Feb 29 in non-leap years)
1774
1754
  # This handles feeds with incorrect dates like "2023-02-29"
1775
- year_match = _RE_FEB29.match(candidate)
1776
- if year_match:
1777
- year = int(year_match.group(1))
1778
- if not ((year % 4 == 0 and year % 100 != 0) or (year % 400 == 0)):
1779
- # Not a leap year, change Feb 29 to Feb 28
1780
- candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
1755
+ if "-02-29" in candidate:
1756
+ year_match = _RE_FEB29.match(candidate)
1757
+ if year_match:
1758
+ year = int(year_match.group(1))
1759
+ if not ((year % 4 == 0 and year % 100 != 0) or (year % 400 == 0)):
1760
+ candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
1781
1761
 
1782
1762
  if "24:00" in candidate:
1783
1763
  candidate = candidate.replace("24:00:00", "00:00:00").replace(
@@ -1786,7 +1766,7 @@ def _parse_date(date_str: str) -> Optional[str]:
1786
1766
 
1787
1767
  dt: Optional[datetime.datetime] = None
1788
1768
 
1789
- is_iso_like = _RE_ISO_LIKE.match(candidate) is not None
1769
+ is_iso_like = len(candidate) >= 10 and candidate[4] == "-" and candidate[0:4].isdigit()
1790
1770
  if is_iso_like:
1791
1771
  iso_candidate = _normalize_iso_datetime_string(candidate)
1792
1772
  try:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.4.9
3
+ Version: 0.5.0
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
File without changes
File without changes