fastfeedparser 0.4.8__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.4.8
3
+ Version: 0.5.0
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = fastfeedparser
3
- version = 0.4.8
3
+ version = 0.5.0
4
4
  author = Vladimir Prelovac
5
5
  author_email = vlad@kagi.com
6
6
  description = High performance RSS, Atom, JSON and RDF feed parser in Python
@@ -1,4 +1,4 @@
1
1
  from .main import parse, FastFeedParserDict
2
2
 
3
- __version__ = "0.4.4"
3
+ __version__ = "0.5.0"
4
4
  __all__ = ["parse", "FastFeedParserDict"]
@@ -3,6 +3,7 @@ from __future__ import annotations
3
3
  import datetime
4
4
  from email.utils import parsedate_to_datetime
5
5
  import gzip
6
+ import html as _html_mod
6
7
  import json
7
8
  import re
8
9
  import zlib
@@ -15,6 +16,7 @@ try:
15
16
  except ImportError:
16
17
  HAS_BROTLI = False
17
18
  from typing import Any, Callable, Optional, TYPE_CHECKING, Literal
19
+ from urllib.parse import urljoin
18
20
  from urllib.request import (
19
21
  HTTPErrorProcessor,
20
22
  HTTPRedirectHandler,
@@ -58,8 +60,8 @@ _RE_UNCLOSED_LINK_BYTES = re.compile(
58
60
  br"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
59
61
  )
60
62
  _RE_FEB29 = re.compile(r"(\d{4})-02-29")
63
+ _RE_HTML_TAGS = re.compile(r"<[^>]+>")
61
64
  _RE_WHITESPACE = re.compile(r"\s+")
62
- _RE_ISO_LIKE = re.compile(r"^\d{4}-\d{2}-\d{2}")
63
65
  _RE_ISO_TZ_NO_COLON = re.compile(r"([+-]\d{2})(\d{2})$")
64
66
  _RE_ISO_TZ_HOUR_ONLY = re.compile(r"([+-]\d{2})$")
65
67
  _RE_ISO_FRACTION = re.compile(r"\.(\d{7,})(?=(?:[+-]\d{2}:?\d{2}|Z|$))", re.IGNORECASE)
@@ -594,7 +596,31 @@ def _raise_for_non_feed_root(
594
596
  raise ValueError(
595
597
  "Received XML sitemap instead of feed (sitemap is for search engines, not a feed)"
596
598
  )
597
- raise ValueError(f"Not a valid feed: {root_tag_local} element found - {error_msg[:100]}")
599
+
600
+
601
+ _RE_META_REFRESH_URL = re.compile(
602
+ r'url\s*=\s*["\']?\s*([^"\'>\s]+)', re.IGNORECASE
603
+ )
604
+
605
+
606
+ def _extract_meta_refresh_url(content: str | bytes, base_url: str) -> str | None:
607
+ """Extract redirect URL from an HTML meta-refresh tag."""
608
+ html_bytes = content.encode("utf-8") if isinstance(content, str) else content
609
+ try:
610
+ doc = etree.fromstring(html_bytes, parser=etree.HTMLParser())
611
+ except Exception:
612
+ return None
613
+ if doc is None:
614
+ return None
615
+
616
+ for meta in doc.iter("meta"):
617
+ if (meta.get("http-equiv") or "").lower() == "refresh":
618
+ match = _RE_META_REFRESH_URL.search(meta.get("content", ""))
619
+ if match:
620
+ url = urljoin(base_url, match.group(1))
621
+ if url != base_url:
622
+ return url
623
+ return None
598
624
 
599
625
 
600
626
  def _detect_feed_structure(
@@ -704,24 +730,26 @@ def _detect_feed_structure(
704
730
  raise ValueError(f"Unknown feed type: {root.tag}")
705
731
 
706
732
 
707
- def parse(source: str | bytes) -> FastFeedParserDict:
708
- """Parse a feed from a URL or XML content.
733
+ def _should_parse_media_content(root: _Element, xml_content: bytes) -> bool:
734
+ """Check if feed likely contains Media RSS fields."""
735
+ ns_values = root.nsmap.values() if root.nsmap else ()
736
+ for ns_value in ns_values:
737
+ if not ns_value:
738
+ continue
739
+ if "search.yahoo.com/mrss" in ns_value:
740
+ return True
709
741
 
710
- Args:
711
- source: URL string or XML content string/bytes
742
+ # Fallback for feeds with undeclared/late namespace usage.
743
+ return b"search.yahoo.com/mrss" in xml_content or b"<media:" in xml_content
712
744
 
713
- Returns:
714
- FastFeedParserDict containing parsed feed data
715
745
 
716
- Raises:
717
- ValueError: If content is empty or invalid
718
- HTTPError: If URL fetch fails
719
- """
720
- if isinstance(source, str) and source.startswith(("http://", "https://")):
721
- xml_content = _fetch_url_content(source)
722
- else:
723
- xml_content = source
746
+ def _should_parse_enclosures(feed_type: _FeedType, xml_content: bytes) -> bool:
747
+ """Check if feed likely contains RSS enclosure elements."""
748
+ return feed_type == "rss" and b"<enclosure" in xml_content
749
+
724
750
 
751
+ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
752
+ """Parse feed content (XML or JSON) that has already been fetched."""
725
753
  json_feed = _maybe_parse_json_feed(xml_content)
726
754
  if json_feed is not None:
727
755
  return json_feed
@@ -734,6 +762,8 @@ def parse(source: str | bytes) -> FastFeedParserDict:
734
762
  feed_type, channel, items, atom_namespace = _detect_feed_structure(
735
763
  root, xml_content, root_tag_local
736
764
  )
765
+ parse_media_content = _should_parse_media_content(root, xml_content)
766
+ parse_enclosures = _should_parse_enclosures(feed_type, xml_content)
737
767
 
738
768
  feed = _parse_feed_info(channel, feed_type, atom_namespace)
739
769
 
@@ -741,7 +771,13 @@ def parse(source: str | bytes) -> FastFeedParserDict:
741
771
  entries: list[FastFeedParserDict] = []
742
772
  feed["entries"] = entries
743
773
  for item in items:
744
- entry = _parse_feed_entry(item, feed_type, atom_namespace)
774
+ entry = _parse_feed_entry(
775
+ item,
776
+ feed_type,
777
+ atom_namespace,
778
+ parse_media_content=parse_media_content,
779
+ parse_enclosures=parse_enclosures,
780
+ )
745
781
  # Ensure that titles and descriptions are always present
746
782
  entry["title"] = entry.get("title", "").strip()
747
783
  entry["description"] = entry.get("description", "").strip()
@@ -750,6 +786,39 @@ def parse(source: str | bytes) -> FastFeedParserDict:
750
786
  return feed
751
787
 
752
788
 
789
+ def parse(source: str | bytes) -> FastFeedParserDict:
790
+ """Parse a feed from a URL or XML content.
791
+
792
+ Args:
793
+ source: URL string or XML content string/bytes
794
+
795
+ Returns:
796
+ FastFeedParserDict containing parsed feed data
797
+
798
+ Raises:
799
+ ValueError: If content is empty or invalid
800
+ HTTPError: If URL fetch fails
801
+ """
802
+ is_url = isinstance(source, str) and source.startswith(("http://", "https://"))
803
+ if is_url:
804
+ content = _fetch_url_content(source)
805
+ else:
806
+ content = source
807
+
808
+ try:
809
+ return _parse_content(content)
810
+ except ValueError as e:
811
+ if not is_url:
812
+ raise
813
+ err_msg = str(e)
814
+ if "HTML" not in err_msg and "not a valid RSS/Atom feed" not in err_msg:
815
+ raise
816
+ redirect_url = _extract_meta_refresh_url(content, source)
817
+ if redirect_url is None:
818
+ raise
819
+ return parse(redirect_url)
820
+
821
+
753
822
  def _parse_feed_info(
754
823
  channel: _Element, feed_type: _FeedType, atom_namespace: Optional[str] = None
755
824
  ) -> FastFeedParserDict:
@@ -1054,16 +1123,9 @@ def _populate_entry_content(
1054
1123
  content_value = entry["content"][0]["value"]
1055
1124
  if content_value:
1056
1125
  if "<" in content_value:
1057
- try:
1058
- html_content = etree.HTML(content_value)
1059
- if html_content is not None:
1060
- content_text = html_content.xpath("string()")
1061
- if isinstance(content_text, str):
1062
- content_value = _RE_WHITESPACE.sub(" ", content_text)
1063
- except etree.ParserError:
1064
- pass
1065
- else:
1066
- content_value = _RE_WHITESPACE.sub(" ", content_value)
1126
+ content_value = _RE_HTML_TAGS.sub(" ", content_value[:2048])
1127
+ content_value = _html_mod.unescape(content_value)
1128
+ content_value = _RE_WHITESPACE.sub(" ", content_value).strip()
1067
1129
  entry["description"] = content_value[:512]
1068
1130
 
1069
1131
 
@@ -1154,12 +1216,133 @@ def _parse_enclosures(item: _Element) -> list[dict[str, Any]] | None:
1154
1216
  return enclosures or None
1155
1217
 
1156
1218
 
1219
+ def _build_rss_item_text_maps(item: _Element) -> tuple[dict[str, Optional[str]], dict[str, Optional[str]]]:
1220
+ by_local: dict[str, Optional[str]] = {}
1221
+ by_full: dict[str, Optional[str]] = {}
1222
+ for child in item:
1223
+ tag = child.tag
1224
+ if not isinstance(tag, str):
1225
+ continue
1226
+ text_value = child.text.strip() if child.text else None
1227
+ if tag not in by_full:
1228
+ by_full[tag] = text_value
1229
+ local = tag.rsplit("}", 1)[-1].lower()
1230
+ if ":" in local:
1231
+ local = local.split(":", 1)[1]
1232
+ if local not in by_local:
1233
+ by_local[local] = text_value
1234
+ return by_local, by_full
1235
+
1236
+
1237
+ def _first_non_empty(mapping: dict[str, Optional[str]], keys: tuple[str, ...]) -> Optional[str]:
1238
+ for key in keys:
1239
+ value = mapping.get(key)
1240
+ if value:
1241
+ return value
1242
+ return None
1243
+
1244
+
1245
+ def _parse_rss_feed_entry_fast(
1246
+ item: _Element,
1247
+ atom_ns: str,
1248
+ parse_media_content: bool = True,
1249
+ parse_enclosures: bool = True,
1250
+ ) -> FastFeedParserDict:
1251
+ text_by_local, text_by_full = _build_rss_item_text_maps(item)
1252
+
1253
+ entry = FastFeedParserDict()
1254
+ atom_id = text_by_full.get(f"{{{atom_ns}}}id")
1255
+ rss_guid = text_by_local.get("guid")
1256
+ rdf_about = item.get("{http://www.w3.org/1999/02/22-rdf-syntax-ns#}about")
1257
+ entry_id: Optional[str] = atom_id or rss_guid or rdf_about
1258
+ if entry_id:
1259
+ entry["id"] = entry_id.strip()
1260
+
1261
+ title = text_by_local.get("title")
1262
+ if title:
1263
+ entry["title"] = title
1264
+
1265
+ description = _first_non_empty(text_by_local, ("description", "summary"))
1266
+ if description:
1267
+ entry["description"] = description
1268
+
1269
+ link = text_by_local.get("link")
1270
+ if link:
1271
+ entry["link"] = link
1272
+
1273
+ published_source = _first_non_empty(text_by_local, ("pubdate", "published", "issued", "date"))
1274
+ if published_source:
1275
+ published = _parse_date(published_source)
1276
+ if published:
1277
+ entry["published"] = published
1278
+
1279
+ updated_source = _first_non_empty(text_by_local, ("lastbuilddate", "updated", "modified"))
1280
+ if updated_source:
1281
+ updated = _parse_date(updated_source)
1282
+ if updated:
1283
+ entry["updated"] = updated
1284
+
1285
+ if "published" not in entry and rss_guid:
1286
+ guid_date = _parse_date(rss_guid)
1287
+ if guid_date:
1288
+ entry["published"] = guid_date
1289
+
1290
+ if "updated" in entry and "published" not in entry:
1291
+ entry["published"] = entry["updated"]
1292
+
1293
+ _populate_entry_links(entry, item, atom_ns)
1294
+ if "id" not in entry and "link" in entry:
1295
+ entry["id"] = entry["link"]
1296
+
1297
+ _populate_entry_content(entry, item, "rss", atom_ns)
1298
+
1299
+ if parse_media_content:
1300
+ media_contents = _parse_media_content(item)
1301
+ if media_contents:
1302
+ entry["media_content"] = media_contents
1303
+
1304
+ if parse_enclosures:
1305
+ enclosures = _parse_enclosures(item)
1306
+ if enclosures:
1307
+ entry["enclosures"] = enclosures
1308
+
1309
+ author = _first_non_empty(text_by_local, ("author", "creator"))
1310
+ if not author:
1311
+ atom_author = item.find(f"{{{atom_ns}}}author/{{{atom_ns}}}name")
1312
+ author = atom_author.text.strip() if atom_author is not None and atom_author.text else None
1313
+ if author:
1314
+ entry["author"] = author
1315
+
1316
+ comments = text_by_local.get("comments")
1317
+ if comments:
1318
+ entry["comments"] = comments
1319
+
1320
+ tags = _parse_tags(item, "rss", atom_ns)
1321
+ if tags:
1322
+ entry["tags"] = tags
1323
+
1324
+ return entry
1325
+
1326
+
1157
1327
  def _parse_feed_entry(
1158
- item: _Element, feed_type: _FeedType, atom_namespace: Optional[str] = None
1328
+ item: _Element,
1329
+ feed_type: _FeedType,
1330
+ atom_namespace: Optional[str] = None,
1331
+ *,
1332
+ parse_media_content: bool = True,
1333
+ parse_enclosures: bool = True,
1159
1334
  ) -> FastFeedParserDict:
1160
1335
  # Use dynamic atom namespace or fallback to default
1161
1336
  atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
1162
1337
 
1338
+ if feed_type == "rss":
1339
+ return _parse_rss_feed_entry_fast(
1340
+ item,
1341
+ atom_ns,
1342
+ parse_media_content=parse_media_content,
1343
+ parse_enclosures=parse_enclosures,
1344
+ )
1345
+
1163
1346
  # Check if this is Atom 0.3 to use different date field names
1164
1347
  is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
1165
1348
 
@@ -1272,38 +1455,27 @@ def _parse_feed_entry(
1272
1455
 
1273
1456
  _populate_entry_content(entry, item, feed_type, atom_ns)
1274
1457
 
1275
- media_contents = _parse_media_content(item)
1276
- if media_contents:
1277
- entry["media_content"] = media_contents
1278
-
1279
- enclosures = _parse_enclosures(item)
1280
- if enclosures:
1281
- entry["enclosures"] = enclosures
1282
-
1283
- author = (
1284
- get_field_value(
1285
- "author",
1286
- f"{{{atom_ns}}}author/{{{atom_ns}}}name",
1287
- "{http://purl.org/dc/elements/1.1/}creator",
1288
- False,
1289
- )
1290
- or get_field_value(
1291
- "{http://purl.org/dc/elements/1.1/}creator",
1292
- "{http://purl.org/dc/elements/1.1/}creator",
1293
- "{http://purl.org/dc/elements/1.1/}creator",
1294
- False,
1295
- )
1296
- or element_get("{http://purl.org/dc/elements/1.1/}creator")
1297
- or element_get("author")
1458
+ if parse_media_content:
1459
+ media_contents = _parse_media_content(item)
1460
+ if media_contents:
1461
+ entry["media_content"] = media_contents
1462
+
1463
+ if parse_enclosures:
1464
+ enclosures = _parse_enclosures(item)
1465
+ if enclosures:
1466
+ entry["enclosures"] = enclosures
1467
+
1468
+ author = get_field_value(
1469
+ "author",
1470
+ f"{{{atom_ns}}}author/{{{atom_ns}}}name",
1471
+ "{http://purl.org/dc/elements/1.1/}creator",
1472
+ False,
1298
1473
  )
1474
+ if not author:
1475
+ author = element_get("{http://purl.org/dc/elements/1.1/}creator") or element_get("author")
1299
1476
  if author:
1300
1477
  entry["author"] = author
1301
1478
 
1302
- if feed_type == "rss":
1303
- comments = element_get("comments")
1304
- if comments:
1305
- entry["comments"] = comments
1306
-
1307
1479
  # Parse entry-level tags/categories
1308
1480
  tags = _parse_tags(item, feed_type, atom_ns)
1309
1481
  if tags:
@@ -1456,7 +1628,7 @@ def _normalize_iso_datetime_string(value: str) -> str:
1456
1628
  if cleaned.endswith(("Z", "z")):
1457
1629
  cleaned = cleaned[:-1] + "+00:00"
1458
1630
 
1459
- if " " in cleaned and "T" not in cleaned[:11] and _RE_ISO_LIKE.match(cleaned):
1631
+ if " " in cleaned and "T" not in cleaned[:11] and len(cleaned) >= 10 and cleaned[4] == "-" and cleaned[0:4].isdigit():
1460
1632
  date_part, rest = cleaned.split(" ", 1)
1461
1633
  if rest and rest[0].isdigit():
1462
1634
  cleaned = f"{date_part}T{rest}"
@@ -1572,18 +1744,20 @@ def _parse_date(date_str: str) -> Optional[str]:
1572
1744
  if not date_str:
1573
1745
  return None
1574
1746
 
1575
- candidate = _RE_WHITESPACE.sub(" ", date_str.strip())
1747
+ candidate = date_str.strip()
1576
1748
  if not candidate:
1577
1749
  return None
1750
+ if "\n" in candidate or "\r" in candidate or "\t" in candidate or " " in candidate:
1751
+ candidate = _RE_WHITESPACE.sub(" ", candidate)
1578
1752
 
1579
1753
  # Fix invalid leap year dates (Feb 29 in non-leap years)
1580
1754
  # This handles feeds with incorrect dates like "2023-02-29"
1581
- year_match = _RE_FEB29.match(candidate)
1582
- if year_match:
1583
- year = int(year_match.group(1))
1584
- if not ((year % 4 == 0 and year % 100 != 0) or (year % 400 == 0)):
1585
- # Not a leap year, change Feb 29 to Feb 28
1586
- candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
1755
+ if "-02-29" in candidate:
1756
+ year_match = _RE_FEB29.match(candidate)
1757
+ if year_match:
1758
+ year = int(year_match.group(1))
1759
+ if not ((year % 4 == 0 and year % 100 != 0) or (year % 400 == 0)):
1760
+ candidate = candidate.replace(f"{year}-02-29", f"{year}-02-28")
1587
1761
 
1588
1762
  if "24:00" in candidate:
1589
1763
  candidate = candidate.replace("24:00:00", "00:00:00").replace(
@@ -1592,7 +1766,8 @@ def _parse_date(date_str: str) -> Optional[str]:
1592
1766
 
1593
1767
  dt: Optional[datetime.datetime] = None
1594
1768
 
1595
- if _RE_ISO_LIKE.match(candidate):
1769
+ is_iso_like = len(candidate) >= 10 and candidate[4] == "-" and candidate[0:4].isdigit()
1770
+ if is_iso_like:
1596
1771
  iso_candidate = _normalize_iso_datetime_string(candidate)
1597
1772
  try:
1598
1773
  dt = datetime.datetime.fromisoformat(iso_candidate)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.4.8
3
+ Version: 0.5.0
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -0,0 +1,53 @@
1
+ from fastfeedparser import parse
2
+ from fastfeedparser.main import _extract_meta_refresh_url
3
+
4
+
5
+ def test_parse_str_with_non_utf8_xml_declaration():
6
+ xml = (
7
+ '<?xml version="1.0" encoding="iso-8859-1"?>'
8
+ '<rss version="2.0">'
9
+ "<channel>"
10
+ "<title>café</title>"
11
+ "<item><title>café</title></item>"
12
+ "</channel>"
13
+ "</rss>"
14
+ )
15
+ feed = parse(xml)
16
+ assert feed.feed.title == "café"
17
+ assert feed.entries[0].title == "café"
18
+
19
+
20
+ def test_parse_bytes_with_non_utf8_encoding():
21
+ xml_bytes = (
22
+ b'<?xml version="1.0" encoding="iso-8859-1"?>'
23
+ b'<rss version="2.0">'
24
+ b"<channel>"
25
+ b"<title>caf\xe9</title>"
26
+ b"<item><title>caf\xe9</title></item>"
27
+ b"</channel>"
28
+ b"</rss>"
29
+ )
30
+ feed = parse(xml_bytes)
31
+ assert feed.feed.title == "café"
32
+ assert feed.entries[0].title == "café"
33
+
34
+
35
+ def test_meta_refresh_extraction():
36
+ html = '<!doctype html><html><head><meta http-equiv=refresh content="0; url=https://example.com/feed.xml"></head></html>'
37
+ assert _extract_meta_refresh_url(html, "https://example.com/feed/") == "https://example.com/feed.xml"
38
+
39
+
40
+ def test_meta_refresh_relative_url():
41
+ html = b'<html><head><meta http-equiv="refresh" content="0;url=/index.xml"></head></html>'
42
+ assert _extract_meta_refresh_url(html, "https://example.com/feed/") == "https://example.com/index.xml"
43
+
44
+
45
+ def test_meta_refresh_none_when_missing():
46
+ html = "<html><head><title>Hello</title></head><body></body></html>"
47
+ assert _extract_meta_refresh_url(html, "https://example.com/") is None
48
+
49
+
50
+ def test_meta_refresh_none_when_same_url():
51
+ html = '<html><head><meta http-equiv="refresh" content="0; url=https://example.com/"></head></html>'
52
+ assert _extract_meta_refresh_url(html, "https://example.com/") is None
53
+
@@ -1,32 +0,0 @@
1
- from fastfeedparser import parse
2
-
3
-
4
- def test_parse_str_with_non_utf8_xml_declaration():
5
- xml = (
6
- '<?xml version="1.0" encoding="iso-8859-1"?>'
7
- '<rss version="2.0">'
8
- "<channel>"
9
- "<title>café</title>"
10
- "<item><title>café</title></item>"
11
- "</channel>"
12
- "</rss>"
13
- )
14
- feed = parse(xml)
15
- assert feed.feed.title == "café"
16
- assert feed.entries[0].title == "café"
17
-
18
-
19
- def test_parse_bytes_with_non_utf8_encoding():
20
- xml_bytes = (
21
- b'<?xml version="1.0" encoding="iso-8859-1"?>'
22
- b'<rss version="2.0">'
23
- b"<channel>"
24
- b"<title>caf\xe9</title>"
25
- b"<item><title>caf\xe9</title></item>"
26
- b"</channel>"
27
- b"</rss>"
28
- )
29
- feed = parse(xml_bytes)
30
- assert feed.feed.title == "café"
31
- assert feed.entries[0].title == "café"
32
-
File without changes
File without changes