fastfeedparser 0.5.8__tar.gz → 0.5.9__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.8
3
+ Version: 0.5.9
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = fastfeedparser
3
- version = 0.5.8
3
+ version = 0.5.9
4
4
  author = Vladimir Prelovac
5
5
  author_email = vlad@kagi.com
6
6
  description = High performance RSS, Atom, JSON and RDF feed parser in Python
@@ -253,12 +253,15 @@ def _fix_malformed_xml_bytes(content: bytes, actual_encoding: str = "utf-8") ->
253
253
  def _prepare_xml_bytes(xml_content: str | bytes) -> bytes:
254
254
  if isinstance(xml_content, bytes):
255
255
  cleaned = _clean_feed_bytes(xml_content)
256
- if not cleaned.strip():
256
+ if not cleaned:
257
257
  raise ValueError("Empty content")
258
258
 
259
259
  # Replace Unicode LINE SEPARATOR (U+2028) and PARAGRAPH SEPARATOR (U+2029)
260
260
  # with regular newlines — these are invalid in XML 1.0 and cause lxml to fail.
261
- if b"\xe2\x80\xa8" in cleaned or b"\xe2\x80\xa9" in cleaned:
261
+ # These are extremely rare; probe a small prefix to avoid full O(n) scan on
262
+ # multi-MB feeds. If neither appears in the first 64 KB, skip the scan.
263
+ _PROBE = cleaned[:65536]
264
+ if b"\xe2\x80\xa8" in _PROBE or b"\xe2\x80\xa9" in _PROBE:
262
265
  cleaned = cleaned.replace(b"\xe2\x80\xa8", b"\n").replace(
263
266
  b"\xe2\x80\xa9", b"\n"
264
267
  )
@@ -519,12 +522,14 @@ _STRICT_XML_PARSER = etree.XMLParser(
519
522
  recover=False,
520
523
  collect_ids=False,
521
524
  resolve_entities=False,
525
+ huge_tree=True,
522
526
  )
523
527
  _RECOVER_XML_PARSER = etree.XMLParser(
524
528
  ns_clean=True,
525
529
  recover=True,
526
530
  collect_ids=False,
527
531
  resolve_entities=False,
532
+ huge_tree=True,
528
533
  )
529
534
 
530
535
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.8
3
+ Version: 0.5.9
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
File without changes
File without changes