fastfeedparser 0.5.0__tar.gz → 0.5.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.0
3
+ Version: 0.5.1
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
@@ -2,3 +2,6 @@
2
2
  requires = ["setuptools~=67.0", "wheel"]
3
3
  build-backend = "setuptools.build_meta"
4
4
 
5
+ [tool.pytest.ini_options]
6
+ testpaths = ["tests"]
7
+
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = fastfeedparser
3
- version = 0.5.0
3
+ version = 0.5.1
4
4
  author = Vladimir Prelovac
5
5
  author_email = vlad@kagi.com
6
6
  description = High performance RSS, Atom, JSON and RDF feed parser in Python
@@ -1,4 +1,4 @@
1
1
  from .main import parse, FastFeedParserDict
2
2
 
3
- __version__ = "0.5.0"
3
+ __version__ = "0.5.1"
4
4
  __all__ = ["parse", "FastFeedParserDict"]
@@ -41,21 +41,12 @@ _RE_XML_DECL_ENCODING = re.compile(
41
41
  _RE_XML_DECL_ENCODING_BYTES = re.compile(
42
42
  br'(<\?xml[^>]*encoding=["\'])([^"\']+)(["\'][^>]*\?>)', re.IGNORECASE
43
43
  )
44
- _RE_DOUBLE_XML_DECL = re.compile(r"<\?xml\?xml\s+", re.IGNORECASE)
45
44
  _RE_DOUBLE_XML_DECL_BYTES = re.compile(br"<\?xml\?xml\s+", re.IGNORECASE)
46
- _RE_DOUBLE_CLOSE = re.compile(r"\?\?>\s*")
47
45
  _RE_DOUBLE_CLOSE_BYTES = re.compile(br"\?\?>\s*")
48
- _RE_UNQUOTED_ATTR = re.compile(r'(\s+[\w:]+)=([^\s>"\']+)')
49
46
  _RE_UNQUOTED_ATTR_BYTES = re.compile(br'(\s+[\w:]+)=([^\s>"\']+)')
50
- _RE_UTF16_ENCODING = re.compile(
51
- r'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
52
- )
53
47
  _RE_UTF16_ENCODING_BYTES = re.compile(
54
48
  br'(<\?xml[^>]*encoding=["\'])utf-16(-le|-be)?(["\'][^>]*\?>)', re.IGNORECASE
55
49
  )
56
- _RE_UNCLOSED_LINK = re.compile(
57
- r"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
58
- )
59
50
  _RE_UNCLOSED_LINK_BYTES = re.compile(
60
51
  br"<link([^>]*[^/])>\s*(?=\n\s*<(?!/link\s*>))", re.MULTILINE
61
52
  )
@@ -112,36 +103,6 @@ def _ensure_utf8_xml_declaration(content: str) -> str:
112
103
  return _RE_XML_DECL_ENCODING.sub(r"\1utf-8\3", content, count=1)
113
104
 
114
105
 
115
- def _clean_feed_text(content: str) -> str:
116
- """Clean feed text by extracting the XML document (if it's embedded in junk)."""
117
- stripped_content = content.lstrip()
118
- stripped_lower = stripped_content[:2000].lower()
119
- if stripped_lower.startswith(("<?xml", "<rss", "<feed", "<rdf")):
120
- return stripped_content
121
-
122
- if stripped_lower.startswith("<!doctype html") or stripped_lower.startswith("<html"):
123
- raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
124
-
125
- xml_start_patterns = (
126
- "<?xml",
127
- "<rss",
128
- "<feed",
129
- "<rdf:rdf",
130
- "<?xml-stylesheet",
131
- )
132
-
133
- content_lines = content.splitlines()
134
- for i, line in enumerate(content_lines):
135
- line_stripped = line.strip().lower()
136
- if any(line_stripped.startswith(pattern) for pattern in xml_start_patterns):
137
- return "\n".join(content_lines[i:])
138
-
139
- if "<script>" in stripped_lower or "<body>" in stripped_lower:
140
- raise ValueError("Content appears to be HTML, not a valid RSS/Atom feed")
141
-
142
- return content
143
-
144
-
145
106
  def _clean_feed_bytes(content: bytes) -> bytes:
146
107
  """Clean feed bytes by extracting the XML document (if it's embedded in junk)."""
147
108
  stripped_content = content.lstrip()
@@ -224,60 +185,9 @@ def _prepare_xml_bytes(xml_content: str | bytes) -> bytes:
224
185
  cleaned = _fix_malformed_xml_bytes(cleaned, actual_encoding=actual_encoding)
225
186
  return cleaned
226
187
 
227
- cleaned_text = _clean_feed_text(xml_content)
228
- if not cleaned_text.strip():
229
- raise ValueError("Empty content")
230
-
231
- needs_fixing = (
232
- "?xml?xml" in cleaned_text[:200]
233
- or "??>" in cleaned_text[:200]
234
- or (
235
- "rss:" in cleaned_text[:500] and "xmlns:rss" not in cleaned_text[:1000]
236
- )
237
- or ("utf-16" in cleaned_text[:200].lower())
238
- )
239
- if needs_fixing:
240
- cleaned_text = _fix_malformed_xml(cleaned_text, actual_encoding="utf-8")
241
-
242
- cleaned_text = _ensure_utf8_xml_declaration(cleaned_text)
243
- return cleaned_text.encode("utf-8", errors="replace")
244
-
245
-
246
- def _fix_malformed_xml(content: str, actual_encoding: str = "utf-8") -> str:
247
- """Fix common malformed XML issues in feeds.
248
-
249
- Some feeds have malformed XML like unclosed link tags or other issues
250
- that can be automatically corrected.
251
-
252
- Args:
253
- content: The XML content as a string
254
- actual_encoding: The actual encoding used (default: utf-8)
255
- """
256
- # Fix double XML declarations like "<?xml?xml version="1.0"?>"
257
- # This is found in dylanharris.org feed
258
- content = _RE_DOUBLE_XML_DECL.sub(r"<?xml ", content)
259
-
260
- # Fix double closing ?> in XML declaration like "??>>"
261
- content = _RE_DOUBLE_CLOSE.sub(r"?>", content)
262
-
263
- # Fix malformed attribute syntax like rss:version=2.0 (missing quotes)
264
- # This is found in dylanharris.org feed
265
- content = _RE_UNQUOTED_ATTR.sub(r'\1="\2"', content)
266
-
267
- # Update encoding in XML declaration to match actual encoding
268
- # This handles cases where content was transcoded from UTF-16 to UTF-8
269
- if actual_encoding.lower() != "utf-16":
270
- content = _RE_UTF16_ENCODING.sub(rf"\1{actual_encoding}\3", content)
271
-
272
- # Fix unclosed link tags - common in Atom feeds
273
- # Pattern: <link ...> followed by whitespace and another tag (not </link>)
274
- # should be <link .../>
275
- # Only fix link tags that are clearly malformed:
276
- # - End with > instead of />
277
- # - Are followed by whitespace and another tag (not a closing </link>)
278
- content = _RE_UNCLOSED_LINK.sub(r"<link\1/>", content)
279
-
280
- return content
188
+ # Str input: fix encoding declaration, encode to bytes, then use bytes path.
189
+ xml_content = _ensure_utf8_xml_declaration(xml_content)
190
+ return _prepare_xml_bytes(xml_content.encode("utf-8", errors="replace"))
281
191
 
282
192
 
283
193
  def _parse_json_feed(json_data: dict) -> FastFeedParserDict:
@@ -556,46 +466,31 @@ def _extract_error_message(root: _Element, raw_bytes: Optional[bytes] = None) ->
556
466
  return error_msg
557
467
 
558
468
 
469
+ _NON_FEED_MESSAGES: dict[str, str] = {
470
+ "html": "Received HTML page instead of feed",
471
+ "div": "Received HTML fragment instead of feed",
472
+ "body": "Received HTML fragment instead of feed",
473
+ "br": "Received HTML fragment instead of feed",
474
+ "status": "Feed server returned status message",
475
+ "error": "Feed server returned error",
476
+ "opml": "Received OPML document instead of feed (OPML is an outline format, not a feed)",
477
+ "urlset": "Received XML sitemap instead of feed (sitemap is for search engines, not a feed)",
478
+ "sitemapindex": "Received XML sitemap instead of feed (sitemap is for search engines, not a feed)",
479
+ }
480
+
481
+
559
482
  def _raise_for_non_feed_root(
560
483
  root: _Element, root_tag_local: str, raw_bytes: Optional[bytes] = None
561
484
  ) -> None:
562
- non_feed_tags = {
563
- "status", "error", "html", "opml", "br", "div", "body",
564
- "urlset", "sitemapindex",
565
- }
566
- if root_tag_local not in non_feed_tags:
485
+ base_msg = _NON_FEED_MESSAGES.get(root_tag_local)
486
+ if base_msg is None:
567
487
  return
568
488
 
569
489
  error_msg = _extract_error_message(root, raw_bytes).strip()[:300] or "No error message"
570
490
 
571
- if root_tag_local == "html":
572
- if error_msg != "No error message" and len(error_msg) > 10:
573
- raise ValueError(f"Received HTML page instead of feed: {error_msg[:150]}")
574
- raise ValueError(
575
- "Received HTML page instead of feed (possible redirect, 404, or server error)"
576
- )
577
- if root_tag_local in {"div", "body"}:
578
- if error_msg != "No error message" and len(error_msg) > 10:
579
- raise ValueError(f"Received HTML fragment instead of feed: {error_msg[:150]}")
580
- raise ValueError("Received HTML fragment instead of feed")
581
- if root_tag_local == "br":
582
- if error_msg != "No error message" and len(error_msg) > 10:
583
- raise ValueError(f"Received HTML error instead of feed: {error_msg[:150]}")
584
- raise ValueError("Received HTML fragment instead of feed")
585
- if root_tag_local == "status":
586
- raise ValueError(f"Feed server returned status message: {error_msg}")
587
- if root_tag_local == "error":
588
- if error_msg != "No error message":
589
- raise ValueError(f"Feed server returned error: {error_msg}")
590
- raise ValueError("Feed server returned error (no details provided)")
591
- if root_tag_local == "opml":
592
- raise ValueError(
593
- "Received OPML document instead of feed (OPML is an outline format, not a feed)"
594
- )
595
- if root_tag_local in {"urlset", "sitemapindex"}:
596
- raise ValueError(
597
- "Received XML sitemap instead of feed (sitemap is for search engines, not a feed)"
598
- )
491
+ if error_msg != "No error message" and len(error_msg) > 10:
492
+ raise ValueError(f"{base_msg}: {error_msg[:150]}")
493
+ raise ValueError(base_msg)
599
494
 
600
495
 
601
496
  _RE_META_REFRESH_URL = re.compile(
@@ -730,24 +625,6 @@ def _detect_feed_structure(
730
625
  raise ValueError(f"Unknown feed type: {root.tag}")
731
626
 
732
627
 
733
- def _should_parse_media_content(root: _Element, xml_content: bytes) -> bool:
734
- """Check if feed likely contains Media RSS fields."""
735
- ns_values = root.nsmap.values() if root.nsmap else ()
736
- for ns_value in ns_values:
737
- if not ns_value:
738
- continue
739
- if "search.yahoo.com/mrss" in ns_value:
740
- return True
741
-
742
- # Fallback for feeds with undeclared/late namespace usage.
743
- return b"search.yahoo.com/mrss" in xml_content or b"<media:" in xml_content
744
-
745
-
746
- def _should_parse_enclosures(feed_type: _FeedType, xml_content: bytes) -> bool:
747
- """Check if feed likely contains RSS enclosure elements."""
748
- return feed_type == "rss" and b"<enclosure" in xml_content
749
-
750
-
751
628
  def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
752
629
  """Parse feed content (XML or JSON) that has already been fetched."""
753
630
  json_feed = _maybe_parse_json_feed(xml_content)
@@ -762,8 +639,6 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
762
639
  feed_type, channel, items, atom_namespace = _detect_feed_structure(
763
640
  root, xml_content, root_tag_local
764
641
  )
765
- parse_media_content = _should_parse_media_content(root, xml_content)
766
- parse_enclosures = _should_parse_enclosures(feed_type, xml_content)
767
642
 
768
643
  feed = _parse_feed_info(channel, feed_type, atom_namespace)
769
644
 
@@ -775,8 +650,6 @@ def _parse_content(xml_content: str | bytes) -> FastFeedParserDict:
775
650
  item,
776
651
  feed_type,
777
652
  atom_namespace,
778
- parse_media_content=parse_media_content,
779
- parse_enclosures=parse_enclosures,
780
653
  )
781
654
  # Ensure that titles and descriptions are always present
782
655
  entry["title"] = entry.get("title", "").strip()
@@ -1245,8 +1118,6 @@ def _first_non_empty(mapping: dict[str, Optional[str]], keys: tuple[str, ...]) -
1245
1118
  def _parse_rss_feed_entry_fast(
1246
1119
  item: _Element,
1247
1120
  atom_ns: str,
1248
- parse_media_content: bool = True,
1249
- parse_enclosures: bool = True,
1250
1121
  ) -> FastFeedParserDict:
1251
1122
  text_by_local, text_by_full = _build_rss_item_text_maps(item)
1252
1123
 
@@ -1282,7 +1153,7 @@ def _parse_rss_feed_entry_fast(
1282
1153
  if updated:
1283
1154
  entry["updated"] = updated
1284
1155
 
1285
- if "published" not in entry and rss_guid:
1156
+ if "published" not in entry and rss_guid and not rss_guid.startswith(("http://", "https://")):
1286
1157
  guid_date = _parse_date(rss_guid)
1287
1158
  if guid_date:
1288
1159
  entry["published"] = guid_date
@@ -1296,15 +1167,13 @@ def _parse_rss_feed_entry_fast(
1296
1167
 
1297
1168
  _populate_entry_content(entry, item, "rss", atom_ns)
1298
1169
 
1299
- if parse_media_content:
1300
- media_contents = _parse_media_content(item)
1301
- if media_contents:
1302
- entry["media_content"] = media_contents
1170
+ media_contents = _parse_media_content(item)
1171
+ if media_contents:
1172
+ entry["media_content"] = media_contents
1303
1173
 
1304
- if parse_enclosures:
1305
- enclosures = _parse_enclosures(item)
1306
- if enclosures:
1307
- entry["enclosures"] = enclosures
1174
+ enclosures = _parse_enclosures(item)
1175
+ if enclosures:
1176
+ entry["enclosures"] = enclosures
1308
1177
 
1309
1178
  author = _first_non_empty(text_by_local, ("author", "creator"))
1310
1179
  if not author:
@@ -1328,20 +1197,12 @@ def _parse_feed_entry(
1328
1197
  item: _Element,
1329
1198
  feed_type: _FeedType,
1330
1199
  atom_namespace: Optional[str] = None,
1331
- *,
1332
- parse_media_content: bool = True,
1333
- parse_enclosures: bool = True,
1334
1200
  ) -> FastFeedParserDict:
1335
1201
  # Use dynamic atom namespace or fallback to default
1336
1202
  atom_ns = atom_namespace or "http://www.w3.org/2005/Atom"
1337
1203
 
1338
1204
  if feed_type == "rss":
1339
- return _parse_rss_feed_entry_fast(
1340
- item,
1341
- atom_ns,
1342
- parse_media_content=parse_media_content,
1343
- parse_enclosures=parse_enclosures,
1344
- )
1205
+ return _parse_rss_feed_entry_fast(item, atom_ns)
1345
1206
 
1346
1207
  # Check if this is Atom 0.3 to use different date field names
1347
1208
  is_atom_03 = atom_ns == "http://purl.org/atom/ns#"
@@ -1434,8 +1295,7 @@ def _parse_feed_entry(
1434
1295
  entry["updated"] = _parse_date(fallback_updated)
1435
1296
 
1436
1297
  # Try to extract date from GUID as final fallback
1437
- if "published" not in entry and rss_guid:
1438
- # Check if GUID contains date information
1298
+ if "published" not in entry and rss_guid and not rss_guid.startswith(("http://", "https://")):
1439
1299
  guid_date = _parse_date(rss_guid)
1440
1300
  if guid_date:
1441
1301
  entry["published"] = guid_date
@@ -1455,15 +1315,13 @@ def _parse_feed_entry(
1455
1315
 
1456
1316
  _populate_entry_content(entry, item, feed_type, atom_ns)
1457
1317
 
1458
- if parse_media_content:
1459
- media_contents = _parse_media_content(item)
1460
- if media_contents:
1461
- entry["media_content"] = media_contents
1318
+ media_contents = _parse_media_content(item)
1319
+ if media_contents:
1320
+ entry["media_content"] = media_contents
1462
1321
 
1463
- if parse_enclosures:
1464
- enclosures = _parse_enclosures(item)
1465
- if enclosures:
1466
- entry["enclosures"] = enclosures
1322
+ enclosures = _parse_enclosures(item)
1323
+ if enclosures:
1324
+ entry["enclosures"] = enclosures
1467
1325
 
1468
1326
  author = get_field_value(
1469
1327
  "author",
@@ -1664,7 +1522,7 @@ def _parsedate_to_utc(value: str) -> Optional[datetime.datetime]:
1664
1522
  return _ensure_utc(parsed)
1665
1523
 
1666
1524
 
1667
- custom_tzinfos: dict[str, int] = {
1525
+ _custom_tzinfos: dict[str, int] = {
1668
1526
  "UTC": 0,
1669
1527
  "UT": 0,
1670
1528
  "GMT": 0,
@@ -1715,7 +1573,7 @@ _DATEPARSER_SETTINGS = {
1715
1573
  @lru_cache(maxsize=512)
1716
1574
  def _slow_dateutil_parse(value: str) -> Optional[datetime.datetime]:
1717
1575
  try:
1718
- return dateutil_parser.parse(value, tzinfos=custom_tzinfos, ignoretz=False)
1576
+ return dateutil_parser.parse(value, tzinfos=_custom_tzinfos, ignoretz=False)
1719
1577
  except (ValueError, TypeError, OverflowError):
1720
1578
  return None
1721
1579
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: fastfeedparser
3
- Version: 0.5.0
3
+ Version: 0.5.1
4
4
  Summary: High performance RSS, Atom, JSON and RDF feed parser in Python
5
5
  Home-page: https://github.com/kagisearch/fastfeedparser
6
6
  Author: Vladimir Prelovac
File without changes
File without changes