fastfeedparser 0.5.7__tar.gz → 0.5.9__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.7
3
+ Version: 0.5.9
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = fastfeedparser
3
- version = 0.5.7
3
+ version = 0.5.9
4
4
  author = Vladimir Prelovac
5
5
  author_email = vlad@kagi.com
6
6
  description = High performance RSS, Atom, JSON and RDF feed parser in Python
@@ -253,12 +253,15 @@ def _fix_malformed_xml_bytes(content: bytes, actual_encoding: str = "utf-8") ->
253
253
  def _prepare_xml_bytes(xml_content: str | bytes) -> bytes:
254
254
  if isinstance(xml_content, bytes):
255
255
  cleaned = _clean_feed_bytes(xml_content)
256
- if not cleaned.strip():
256
+ if not cleaned:
257
257
  raise ValueError("Empty content")
258
258
 
259
259
  # Replace Unicode LINE SEPARATOR (U+2028) and PARAGRAPH SEPARATOR (U+2029)
260
260
  # with regular newlines — these are invalid in XML 1.0 and cause lxml to fail.
261
- if b"\xe2\x80\xa8" in cleaned or b"\xe2\x80\xa9" in cleaned:
261
+ # These are extremely rare; probe a small prefix to avoid full O(n) scan on
262
+ # multi-MB feeds. If neither appears in the first 64 KB, skip the scan.
263
+ _PROBE = cleaned[:65536]
264
+ if b"\xe2\x80\xa8" in _PROBE or b"\xe2\x80\xa9" in _PROBE:
262
265
  cleaned = cleaned.replace(b"\xe2\x80\xa8", b"\n").replace(
263
266
  b"\xe2\x80\xa9", b"\n"
264
267
  )
@@ -519,12 +522,14 @@ _STRICT_XML_PARSER = etree.XMLParser(
519
522
  recover=False,
520
523
  collect_ids=False,
521
524
  resolve_entities=False,
525
+ huge_tree=True,
522
526
  )
523
527
  _RECOVER_XML_PARSER = etree.XMLParser(
524
528
  ns_clean=True,
525
529
  recover=True,
526
530
  collect_ids=False,
527
531
  resolve_entities=False,
532
+ huge_tree=True,
528
533
  )
529
534
 
530
535
 
@@ -991,8 +996,19 @@ def _parse_feed_info(
991
996
  for link in channel.findall(f"{{{atom_ns}}}link"):
992
997
  rel = link.get("rel")
993
998
  href = link.get("href") or link.get("link")
994
- if rel is None and href:
999
+ if rel == "alternate" and href and not feed_link:
995
1000
  feed_link = href
1001
+ feed_links.append(
1002
+ {
1003
+ "rel": rel,
1004
+ "type": link.get("type"),
1005
+ "href": href,
1006
+ "title": link.get("title"),
1007
+ }
1008
+ )
1009
+ elif rel is None and href:
1010
+ if not feed_link:
1011
+ feed_link = href
996
1012
  elif rel not in {"hub", "self", "replies", "edit"}:
997
1013
  feed_links.append(
998
1014
  {
@@ -1134,7 +1150,10 @@ def _populate_entry_links_from_elements(
1134
1150
  "title": link.get("title"),
1135
1151
  }
1136
1152
  if rel == "alternate":
1137
- alternate_link = link_dict
1153
+ if alternate_link is None:
1154
+ alternate_link = link_dict
1155
+ else:
1156
+ entry_links.append(link_dict)
1138
1157
  elif rel not in {"edit", "self"}:
1139
1158
  entry_links.append(link_dict)
1140
1159
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.7
3
+ Version: 0.5.9
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
File without changes
File without changes