fastfeedparser 0.5.8__tar.gz → 0.5.10__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.8
3
+ Version: 0.5.10
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = fastfeedparser
3
- version = 0.5.8
3
+ version = 0.5.10
4
4
  author = Vladimir Prelovac
5
5
  author_email = vlad@kagi.com
6
6
  description = High performance RSS, Atom, JSON and RDF feed parser in Python
@@ -253,12 +253,15 @@ def _fix_malformed_xml_bytes(content: bytes, actual_encoding: str = "utf-8") ->
253
253
  def _prepare_xml_bytes(xml_content: str | bytes) -> bytes:
254
254
  if isinstance(xml_content, bytes):
255
255
  cleaned = _clean_feed_bytes(xml_content)
256
- if not cleaned.strip():
256
+ if not cleaned:
257
257
  raise ValueError("Empty content")
258
258
 
259
259
  # Replace Unicode LINE SEPARATOR (U+2028) and PARAGRAPH SEPARATOR (U+2029)
260
260
  # with regular newlines — these are invalid in XML 1.0 and cause lxml to fail.
261
- if b"\xe2\x80\xa8" in cleaned or b"\xe2\x80\xa9" in cleaned:
261
+ # These are extremely rare; probe a small prefix to avoid full O(n) scan on
262
+ # multi-MB feeds. If neither appears in the first 64 KB, skip the scan.
263
+ _PROBE = cleaned[:65536]
264
+ if b"\xe2\x80\xa8" in _PROBE or b"\xe2\x80\xa9" in _PROBE:
262
265
  cleaned = cleaned.replace(b"\xe2\x80\xa8", b"\n").replace(
263
266
  b"\xe2\x80\xa9", b"\n"
264
267
  )
@@ -519,12 +522,14 @@ _STRICT_XML_PARSER = etree.XMLParser(
519
522
  recover=False,
520
523
  collect_ids=False,
521
524
  resolve_entities=False,
525
+ huge_tree=True,
522
526
  )
523
527
  _RECOVER_XML_PARSER = etree.XMLParser(
524
528
  ns_clean=True,
525
529
  recover=True,
526
530
  collect_ids=False,
527
531
  resolve_entities=False,
532
+ huge_tree=True,
528
533
  )
529
534
 
530
535
 
@@ -630,6 +635,7 @@ def _raise_for_non_feed_root(
630
635
 
631
636
 
632
637
  _RE_META_REFRESH_URL = re.compile(r'url\s*=\s*["\']?\s*([^"\'>\s]+)', re.IGNORECASE)
638
+ _MAX_META_REDIRECTS = 3
633
639
 
634
640
 
635
641
  def _extract_meta_refresh_url(content: str | bytes, base_url: str) -> str | None:
@@ -861,31 +867,32 @@ def parse(
861
867
  else:
862
868
  content = source
863
869
 
864
- try:
865
- return _parse_content(
866
- content,
867
- include_content=include_content,
868
- include_tags=include_tags,
869
- include_media=include_media,
870
- include_enclosures=include_enclosures,
871
- )
872
- except ValueError as e:
873
- if not is_url:
874
- raise
875
- assert isinstance(source, str)
876
- err_msg = str(e)
877
- if "HTML" not in err_msg and "not a valid RSS/Atom feed" not in err_msg:
878
- raise
879
- redirect_url = _extract_meta_refresh_url(content, source)
880
- if redirect_url is None:
881
- raise
882
- return parse(
883
- redirect_url,
884
- include_content=include_content,
885
- include_tags=include_tags,
886
- include_media=include_media,
887
- include_enclosures=include_enclosures,
888
- )
870
+ parse_kwargs = dict(
871
+ include_content=include_content,
872
+ include_tags=include_tags,
873
+ include_media=include_media,
874
+ include_enclosures=include_enclosures,
875
+ )
876
+
877
+ redirects_left = _MAX_META_REDIRECTS
878
+ while True:
879
+ try:
880
+ return _parse_content(content, **parse_kwargs)
881
+ except ValueError as e:
882
+ if not is_url:
883
+ raise
884
+ assert isinstance(source, str)
885
+ err_msg = str(e)
886
+ if "HTML" not in err_msg and "not a valid RSS/Atom feed" not in err_msg:
887
+ raise
888
+ if redirects_left <= 0:
889
+ raise ValueError("too many meta-refresh redirects") from e
890
+ redirect_url = _extract_meta_refresh_url(content, source)
891
+ if redirect_url is None:
892
+ raise
893
+ content = _fetch_url_content(redirect_url)
894
+ source = redirect_url
895
+ redirects_left -= 1
889
896
 
890
897
 
891
898
  def _parse_feed_info(
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.8
3
+ Version: 0.5.10
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
File without changes